diff --git a/.claude-plugin/plugin.json b/.claude-plugin/plugin.json index 48b905b..fa3e904 100644 --- a/.claude-plugin/plugin.json +++ b/.claude-plugin/plugin.json @@ -12,7 +12,9 @@ "./skills/flowx-setup", "./skills/flowx-discover", "./skills/flowx-enrich", + "./skills/flowx-route", "./skills/flowx-convert", + "./skills/flowx-resolve-airflow-gaps", "./skills/flowx-package", "./skills/flowx-migrate" ] diff --git a/AGENTS.md b/AGENTS.md index e1ae8c4..2a48b9e 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -73,10 +73,13 @@ intermediates under `.work/` (pruned by `package`). | `models/adf_ast.py` | Typed AST nodes for ADF definitions | | `models/ir.py` | Databricks intermediate representation | | `models/dab.py` | DAB output schema types | -| `parser/adf_loader.py` | Parses ADF exports, produces `metadata/inventory.json` + `metadata/profile_report.csv` | +| `models/discovery.py` | Source-faithful shared discovery AST (`SourceGraph`) that both ADF and Airflow map onto | +| `sources/adf/loader.py` | Parses ADF exports, produces `metadata/inventory.json` + `metadata/profile_report.csv` | +| `sources/adf/discovery_mapping.py` | Maps the ADF AST onto the shared `SourceGraph` discovery model (1:1, lossless) | +| `sources/adf/translate.py` | Registry dispatch, topological sort, context threading | +| `sources/adf/translators/` | One module per deterministic activity type (16 total) | +| `sources/airflow/` | Airflow source: loader, discover, and convert (mirrors the ADF source layout) | | `parser/expression_parser.py` | Translates ADF expressions (@activity, @pipeline, @variables) | -| `translator/engine.py` | Registry dispatch, topological sort, context threading | -| `translator/activity_translators/` | One module per deterministic activity type (16 total) | | `preparer/workflow_preparer.py` | Orchestrates activity preparers | | `preparer/code_generator.py` | Notebook code generation for activity types | | `preparer/activity_preparers/` | One module per activity type | @@ -86,6 +89,9 @@ intermediates under `.work/` (pruned by `package`). | `reporting/coverage.py` | Builds per-pipeline coverage rows from `metadata/` | | `reporting/results.py` | Writes per-run coverage to a UC table (run_id/run_date/run_by) via the SDK | | `reporting/dashboard.py` | Installs + publishes an AI/BI coverage dashboard over the results table | +| `routing.py` | Groups pipelines into connected components over control lineage; recommends deterministic/agentic per component (both options) and records the user's decision as `metadata/conversion_plan.json` (additive; convert untouched) | +| `route_agentic.py` | Applies a routing decision after convert: rewrites the translation report for agentic-routed groups (placeholder tasks + per-task `AgenticGap`), keeping the fill in-engine | +| `models/conversion_plan.py` | Source-neutral conversion-plan artifact model (per-component decision + both options), the routing counterpart to `models/insights.py` | ## Activity Types @@ -127,11 +133,11 @@ ExecuteDataFlow, SqlServerStoredProcedure, AzureFunction, WebHook, Custom, Execu ## Adding a New Deterministic Translator 1. Add IR dataclass to `src/flowx/models/ir.py` -2. Create translator at `src/flowx/translator/activity_translators/.py` +2. Create translator at `src/flowx/sources/adf/translators/.py` 3. Create preparer at `src/flowx/preparer/activity_preparers/.py` 4. Add notebook generator to `src/flowx/preparer/code_generator.py` if needed -5. Register in engine.py (TRANSLATOR_REGISTRY for leaf, match statement for control-flow) -6. Move from AGENTIC_TYPES to DETERMINISTIC_TYPES in adf_loader.py +5. Register in `src/flowx/sources/adf/translate.py` (TRANSLATOR_REGISTRY for leaf, match statement for control-flow) +6. Move from AGENTIC_TYPES to DETERMINISTIC_TYPES in `src/flowx/sources/adf/loader.py` 7. Update activity-mapping.md reference 8. Add test fixtures and unit tests @@ -153,3 +159,10 @@ only one-line pointers): proxy the SDK sees the workspace `Origin` and a proxied `Host: localhost:`, so its Host/Origin allowlist misfires (403/421) while adding nothing on top of the proxy's authentication. Browser CORS is a separate concern configured via `FLOWX_ALLOWED_ORIGINS`. + +## Maintaining this file + +Keep this file for knowledge useful to almost every future agent session in this project. +Do not repeat what the codebase already shows; point to the authoritative file or command instead. +Prefer rewriting or pruning existing entries over appending new ones. +When updating this file, preserve this bar for all agents and keep entries concise. diff --git a/CLAUDE.md b/CLAUDE.md deleted file mode 100644 index 43c994c..0000000 --- a/CLAUDE.md +++ /dev/null @@ -1 +0,0 @@ -@AGENTS.md diff --git a/CLAUDE.md b/CLAUDE.md new file mode 120000 index 0000000..47dc3e3 --- /dev/null +++ b/CLAUDE.md @@ -0,0 +1 @@ +AGENTS.md \ No newline at end of file diff --git a/skills/flowx-migrate/SKILL.md b/skills/flowx-migrate/SKILL.md index e12b05c..aa1567a 100644 --- a/skills/flowx-migrate/SKILL.md +++ b/skills/flowx-migrate/SKILL.md @@ -2,7 +2,8 @@ name: flowx-migrate description: > End-to-end migration of a source orchestrator's pipelines (Azure Data Factory, Apache Airflow) - to Databricks Lakeflow Jobs. Orchestrates discover, convert, and package phases in sequence. + to Databricks Lakeflow Jobs. Orchestrates discover → enrich → convert (deterministic baseline) → + route → fill → package in sequence. triggers: - "migrate pipelines" - "migrate ADF" @@ -16,16 +17,32 @@ triggers: # End-to-End Source to Databricks Migration Orchestrate the complete migration of a source orchestrator's pipelines to Databricks Lakeflow Jobs -via Declarative Automation Bundles. This skill runs all three phases in sequence: discover, convert, -package. +via Declarative Automation Bundles. This skill runs the full flow in sequence: discover → enrich +(default) → convert (deterministic baseline) → route (decide + edit) → fill agentic gaps → package. ## Context This is the top-level orchestration skill. It runs the full migration pipeline: +> **Scope:** deterministic discover → convert → package works for **both ADF and Airflow**. The +> **routing + agentic-conversion path (enrich → route → fill)** applies to **ADF today** — Airflow +> discovery does not yet emit control lineage or motifs, so it cannot form routing components, and +> `convert --merge-agentic --source airflow` is disabled. Airflow follows its own +> `flowx-resolve-airflow-gaps` path and aligns with routing via #63. + 1. **Discover** — Parse the source's definitions into a typed inventory -2. **Convert** — Convert the source's tasks to Databricks IR (deterministic + agentic) -3. **Package** — Generate Databricks Declarative Automation Bundles for deployment +2. **Enrich** *(default)* — Author the agentic `insights` layer over the inventory and merge it + (`flowx-enrich`); skippable for a deterministic-only pass +3. **Convert (deterministic baseline)** — Convert the source's tasks to Databricks IR, producing + `.work/translation_report.json` (`flowx-convert`). Runs **before** routing +4. **Route** — Decide, per connected component, deterministic vs. agentic conversion; record the + fingerprint-bound `metadata/conversion_plan.json`; edit the baseline report so routed-agentic + groups become placeholder gaps (`flowx-route`) +5. **Fill agentic gaps** — Replace the routed-agentic placeholders: per-pipeline + `convert --merge-agentic` (ADF only) or cross-pipeline `fill-agentic combine`; Airflow gaps go + through `flowx-resolve-airflow-gaps`. **Never re-run a plain `convert` after routing** — it + rewrites the report and erases the placeholders +6. **Package** — Generate Databricks Declarative Automation Bundles for deployment Each phase builds on the output of the previous phase. The user is shown a summary and asked to confirm before proceeding to the next phase. @@ -40,7 +57,7 @@ source path (`--adf-source-path` / `--airflow-source-path`, both aliases of `--s ## How to run this skill — MCP tools or venv CLI -This skill orchestrates all three phases. Run the **`setup`** skill first if you haven't. There are +This skill orchestrates the full flow. Run the **`setup`** skill first if you haven't. There are two execution paths: ### MCP tools (Databricks Genie Code, or a local stdio registration) @@ -101,8 +118,11 @@ discover/convert and for `inputs discover`/`inputs convert`; for Airflow, swap ` ``` flowx(command="inputs", parameters={"phase": "discover", "source": "adf"}) # source req for discover/convert flowx(command="discover", parameters={"source": "adf", "adf_definitions": {...}, "output_dir": ..., "pipeline": ...}) -flowx(command="convert", parameters={"source": "adf", "output_dir": ..., "pipeline": ...}) -flowx(command="merge_agentic", parameters={"source": "adf", "report_path": ..., "agentic_results_dir": ..., "output_path": ...}) # ADF only, if agentic results +flowx(command="enrich", parameters={"output_dir": ..., "insights": {...}}) # default: author + merge the insights layer (see flowx-enrich) +flowx(command="convert", parameters={"source": "adf", "output_dir": ..., "pipeline": ...}) # deterministic baseline, BEFORE route +flowx(command="route", parameters={"output_dir": ...}) # recommend; re-call with "plan": {...} to record + edit the report (see flowx-route) +flowx(command="merge_agentic", parameters={"source": "adf", "report_path": ..., "agentic_results_dir": ..., "output_path": ...}) # ADF only, per-pipeline agentic fill +flowx(command="fill_agentic", parameters={"output_dir": ..., "members": [...], "pipelines": [...]}) # cross-pipeline combine of a routed-agentic group flowx(command="inspect", parameters={"report_path": ...}) flowx(command="apply_answers", parameters={"report_path": ..., "answers": [...], "output_dir": ...}) flowx(command="package", parameters={"output_dir": ..., "catalog": ..., "schema": ...}) @@ -209,15 +229,35 @@ If the user says no, explain the options: - Review `/metadata/inventory.json` (and `profile_report.csv`) to understand unsupported activities and pipeline complexity - Manually classify activities before proceeding -If the user says yes, proceed to step 4. +If the user says yes, proceed to Step 4. + +### Step 4 — Enrich the inventory (default) + +By default, chain into enrichment: invoke the **`flowx:flowx-enrich`** skill to author the agentic +`insights` layer (factory-wide recommendation, per-pipeline intent + recommended Databricks patterns, +cross-pipeline relationships) and merge it into `/metadata/inventory.json`. flowx has no +LLM — you author the insights JSON and `enrich` validates + merges it additively, leaving every +deterministic inventory key byte-identical. The routing step consumes this block to present the +agentic conversion option per group. -### Step 4 — Phase 2: Convert +**Skip only for a deterministic-only pass.** When the user explicitly asked for a headless, +deterministic-only migration (no agent/LLM authoring), skip enrich — the deterministic +`inventory.json` is complete and `flowx-route` still works from the structure alone. Otherwise enrich +by default. -Invoke the `flowx:flowx-convert` skill with: +### Step 5 — Phase 2: Convert (deterministic baseline) + +Convert the source's tasks to Databricks IR, producing the **deterministic baseline** report +`/.work/translation_report.json`. This runs **before** routing, because routing (Step 7) +reads and edits this report. Invoke the `flowx:flowx-convert` skill with: - `--source `: the same source discover used - Source path: the original source path (same one discover used) - Output dir: the same shared `` (convert writes its report to `/.work/`) +Run this deterministic convert **exactly once, here, before routing.** The only convert *after* +routing is the additive `convert --merge-agentic` fill in Step 8 — a second plain `convert` would +rewrite `.work/translation_report.json` and erase the routed-agentic placeholders. + Wait for the translation to complete and present the summary: ``` @@ -229,7 +269,7 @@ Failed: 4 ( 8.5%) Overall coverage: 91.5% ``` -### Step 5 — Present translation details +### Step 6 — Present translation details Show the user: 1. What was translated deterministically (bulk — just counts by type) @@ -241,7 +281,51 @@ For failures, suggest: - Retry with additional context - Skip and add placeholder -### Step 5.1 — Gather just-in-time translation configuration +### Step 7 — Route the conversion (deterministic vs. agentic) + +**ADF only today.** Routing (and the agentic fill in Step 8) applies to **ADF**. For **Airflow**, skip +Steps 7–8 and take the deterministic report straight to configuration/package; resolve any Airflow +agentic gaps with the **`flowx-resolve-airflow-gaps`** skill instead (Airflow discovery does not yet +emit control lineage or motifs — routing aligns via #63). + +Routing reads the deterministic baseline report `.work/translation_report.json` (built in Step 5) and +edits it. Invoke the **`flowx:flowx-route`** skill to present the per-connected-component +recommendation, take the **customer's** decision (interactive, or an authored plan — the customer +decides deterministic-vs-agentic per component; the agent presents the options and serializes only the +approved decision), and record the fingerprint-bound `/metadata/conversion_plan.json`. +Recording a plan edits `.work/translation_report.json` so every routed-**agentic** group's tasks +become placeholder gaps; deterministic groups (and a no-agentic-route plan) leave the report +untouched — non-breaking. + +If Step 5 was skipped and the report is missing, `route` can trigger the convert phase in-process +once (pass `--source` / `--source-path`); it never runs a second plain convert once a report exists. +If the user wants a straight deterministic migration, they accept the all-deterministic recommendation +here and the rest of the flow is unchanged (no placeholders, so Step 8 is a no-op). + +### Step 8 — Fill routed-agentic gaps + +If Step 7 routed any component **agentic**, its pipelines' tasks are now `PlaceholderActivity` nodes +with one tagged gap each. Author the replacements and merge them (no LLM in the library — it validates +and merges) **before** the just-in-time config below, so the filled tasks get configuration-stamped +and packaged: + +- **Per-pipeline** (the pipeline stays 1:1, **ADF only**): author one result JSON per gap and merge + with `convert --source adf --merge-agentic --report .work/translation_report.json --agentic-results ` + (`--source adf` is mandatory; the convert phase exits 2 without it). The merge replaces, per result, + the **first task whose `name` matches `activity_name`** — it matches by name and does **not** check + the target is a placeholder, so make sure each name targets the intended agentic gap. Airflow's + `--merge-agentic` is disabled (exits 2) — for Airflow per-gap fills use the + **`flowx-resolve-airflow-gaps`** skill instead. +- **Cross-pipeline COMBINE** (N pipelines → M, e.g. one Lakeflow Connect pipeline): author the + replacement pipeline IR (typically with `AgenticComponentActivity` nodes) and run + `fill-agentic combine --output-dir --members "" --pipelines-path `; + the members must exactly match the routed-agentic component and the merged report is validated + structurally before it is written. + +Skip this step entirely when no component was routed agentic. See the **`flowx-route`** skill for the +full fill details and the `AgenticComponentActivity` shape. + +### Step 9 — Gather just-in-time translation configuration Run `inspect` **once** to get the full option schema (every option carries a `show_when` condition), then drive the chain locally — ask an option only when its `show_when` clauses are all satisfied by @@ -296,7 +380,7 @@ incremental_load_watermark, metadata_driven_bulk_copy, ...) the adapter emits on `consolidate_motif:` option. The user must explicitly opt in to `consolidate` for each detected pattern. -### Step 6 — Checkpoint: confirm proceed to bundle generation +### Step 10 — Checkpoint: confirm proceed to bundle generation > Translation is 91.5% complete. 4 activities could not be translated automatically. > Options: @@ -306,7 +390,7 @@ for each detected pattern. > > What would you like to do? -### Step 6.5 — Detect workspace artifacts and authenticate +### Step 11 — Detect workspace artifacts and authenticate Before invoking the package phase, run the adapter's `workspace-paths` subcommand to detect any absolute workspace paths @@ -326,14 +410,14 @@ When the response carries `needs_auth: true`: services in the ADF export). 2. Run `databricks auth login --host ` interactively to set up a local profile. -3. Pass `--profile ` to the prepare invocation in Step 7 so +3. Pass `--profile ` to the prepare invocation in Step 12 so flowx downloads the referenced notebooks and downloads them under `bundle/src/notebooks/` with the task references rewritten to the relative `../src/notebooks/...` paths. Skip this step entirely when `needs_auth` is `false`. -### Step 7 — Phase 3: Package +### Step 12 — Phase 3: Package Invoke the `flowx:flowx-package` skill with: - Output dir: the same shared `` — package reads the stamped report from @@ -344,7 +428,7 @@ Invoke the `flowx:flowx-package` skill with: Package prunes the transient `/.work/` after a successful build, leaving the bundle (databricks.yml, resources/, src/, SETUP.md) plus the kept `metadata/` folder. -### Step 7.5 — (Optional) Persist coverage results and install a dashboard +### Step 13 — (Optional) Persist coverage results and install a dashboard When running with workspace auth (Genie Code or a configured profile), offer to record this run's migration coverage to a Unity Catalog table and optionally install a coverage dashboard. @@ -370,7 +454,7 @@ Creates and publishes an AI/BI coverage dashboard over the table and prints its auto-detect a SQL warehouse when `--warehouse-id` is omitted and degrade gracefully without workspace auth. See the `package` skill (Step 8) for details. -### Step 8 — Present final summary +### Step 14 — Present final summary Display the complete migration summary: @@ -409,7 +493,7 @@ Next Steps: 7. Verify job output and promote to staging/prod ``` -### Step 9 — Offer follow-up actions +### Step 15 — Offer follow-up actions Ask if the user wants to: 1. Validate the bundle now (`databricks bundle validate`) @@ -432,7 +516,7 @@ See `references/workflow.md` for a detailed description of the three-phase archi ## Output Artifacts -All three phases write into a single shared `` (default `./flowx_output`): +All phases write into a single shared `` (default `./flowx_output`): | Path | Phase | Contents | |---|---|---| diff --git a/skills/flowx-migrate/references/workflow.md b/skills/flowx-migrate/references/workflow.md index faf083b..cb345ef 100644 --- a/skills/flowx-migrate/references/workflow.md +++ b/skills/flowx-migrate/references/workflow.md @@ -60,6 +60,28 @@ ADF JSON Exports - Datasets and linked services are parsed for context but not independently translated — they inform the activity translators. - Triggers are included in the inventory and translated in phase 2. +## Between Discover and Convert: Enrich (default) + Route + +**Skills:** `flowx:flowx-enrich`, `flowx:flowx-route` + +After discover writes the deterministic inventory, the standard flow enriches and routes before +convert. Both are additive and contain **no LLM** — the agent authors, the library validates and +merges: + +- **Enrich (default):** the agent authors an `insights` layer (factory-wide recommendation, + per-pipeline intent + recommended Databricks patterns, cross-pipeline relationships) and `enrich` + merges it into `metadata/inventory.json` under a single additive `insights` key, leaving every + deterministic key byte-identical. Skippable for a deterministic-only, headless pass. +- **Route:** groups pipelines into connected components over control lineage and, per component, + records a `deterministic` or `agentic` decision as the fingerprint-bound + `metadata/conversion_plan.json`. Recording an agentic decision edits `.work/translation_report.json` + so those groups' tasks become placeholder gaps; a fully-deterministic plan (or no plan) leaves + convert/package behaving exactly as before — the non-breaking guarantee. + +Routed-agentic groups are then filled during convert: per-pipeline via `convert --merge-agentic`, or +cross-pipeline (N→M, e.g. a Lakeflow Connect collapse) via `fill-agentic combine` using authored +`AgenticComponentActivity` nodes. + ## Phase 2: Convert **Skill:** `flowx:flowx-convert` diff --git a/skills/flowx-route/SKILL.md b/skills/flowx-route/SKILL.md new file mode 100644 index 0000000..70fdd13 --- /dev/null +++ b/skills/flowx-route/SKILL.md @@ -0,0 +1,311 @@ +--- +name: flowx-route +description: > + Route each connected component of the discovered inventory to a deterministic (1:1 engine) or + agentic (LLM-assisted re-architecture) conversion, record the fingerprint-bound conversion plan, + and fill the routed-agentic groups — per-pipeline via convert --merge-agentic, or cross-pipeline + via fill-agentic combine. Runs after enrich, before/with convert. +triggers: + - "route pipelines" + - "route conversion" + - "conversion plan" + - "deterministic or agentic" + - "fill agentic" + - "combine pipelines" + - "agentic conversion" + - "recommend conversion route" +--- + +# Route the Conversion (deterministic vs. agentic) and Fill the Gaps + +After discover (and, by default, `flowx-enrich`), routing decides **per connected component** whether +each part of the factory converts **deterministically** (the typed engine's 1:1 translation) or +**agentically** (an LLM-authored re-architecture, e.g. collapsing five extractors onto one Lakeflow +Connect pipeline). It records the decision as a fingerprint-bound `metadata/conversion_plan.json`, +edits the translation report so routed-agentic groups become placeholder gaps, and then you author +the fill. + +**ADF only (current scope).** The discover → enrich → route → edit → fill agentic-conversion flow is +supported for **ADF today**. Airflow is **not** yet wired for routing: its discovery does not emit +control lineage or motifs, so it cannot form multi-DAG routing components, and `convert +--merge-agentic --source airflow` is disabled. Airflow aligns with routing once #63 maps it onto the +shared discovery AST. For **Airflow agentic gaps today, use the `flowx-resolve-airflow-gaps` skill** +(the strict per-gap resolver) — not this routing flow. + +**There is no LLM inside flowx.** The library computes the recommendation deterministically and only +*validates and records* the decision and the authored fill — the same author → validate → merge +contract `enrich` uses. This is additive and non-breaking: with **no recorded plan**, `convert` and +`package` behave exactly as before. + +**Who decides what.** The library **recommends** a route per component; the **customer decides** +deterministic-vs-agentic for each component; the **agent** presents the options and the +recommendation, serializes only the customer's *approved* decision into the plan, and authors the +agentic fill. The agent never picks the route on the customer's behalf — it records the customer's +choice and does the mechanical work of the approved fill. + +## Where routing sits + +``` +discover → enrich (default) → convert (deterministic baseline) → route (decide + edit) → fill-agentic → package +``` + +Convert builds the deterministic baseline report **before** routing (route can trigger it in-process +via `--source` / `--source-path`); routing then edits that report. After routing has recorded agentic +placeholders, the only convert is the additive `convert --merge-agentic` fill — never a second plain +`convert`, which would overwrite the report and erase the placeholders. + +Pipelines are grouped into weak/undirected **connected components** over the inventory's control +lineage (`lineage.control_edges`), so mutually-referencing pipelines are decided together and a +caller/callee reference is never split across incompatible routes. ADF emits these control edges +(from `ExecutePipeline`) in discovery; Airflow discovery does not emit control lineage yet, which is +why routing is ADF-only today (see the scope note above). + +Run the **`setup`** skill first if you haven't. Everything below has an MCP-tool path (Genie Code, or +a local stdio registration — call the single **`flowx`** tool, run no `python3`/`$PY`) and a venv-CLI +path (local; `PY="$(cat /.migration-venv)"` and `export PYTHONPATH="/src"`). + +## Step 1 — Recommend (the dry run you read first) + +Call `route` with **no decision** to get the recommendation: every component with its `members`, both +first-class conversion `options` (a *deterministic* option carrying the engine-capability assessment ++ any uncovered gaps, and an *agentic* option carrying the `recommended_patterns` from `enrich`, with +any `simplification_pattern` flagged), the `findings` (unresolved/dangling control edges kept, never +severed), and a ready-to-record `default_plan` proposing `decision == recommended` for every +component. + +- **MCP tool:** `flowx(command="route", parameters={"output_dir": ""})` +- **venv CLI:** + + ```bash + export PYTHONPATH="/src" + PY="$(cat /.migration-venv)" + "$PY" -m flowx.adapter route --output-dir + ``` + + On a non-TTY with no `--plan-path`, this emits the recommendation and exits 0 (it edits nothing). + `route` reads `metadata/inventory.json`; run discover first. `--out ` writes the JSON to a + file instead of stdout. + +`recommended` is the library's starting suggestion — `deterministic` when the whole component is +engine-capable, else `agentic`. Present each component's members, its recommendation, and the agentic +option's patterns (flag any `has_simplification` re-architecture prominently). + +## Step 2 — Decide and record the plan + +Take the per-component decision and record it. `route` validates the plan, writes the fingerprint-bound +`metadata/conversion_plan.json`, and then edits `.work/translation_report.json` + `gaps.json` so every +routed-**agentic** group's tasks become placeholder gaps (deterministic groups are left byte-identical; +when nothing is routed agentic the report is untouched — the non-breaking guarantee). + +There are three ways to supply the decision: + +- **Interactive prompt (TTY).** Run `route` with no `--plan-path` on a real terminal and it asks, per + component, `route [d]eterministic / [a]gentic (default=)`. An empty answer accepts the + recommendation. This is the from-the-seat path. +- **Authored plan file** — `--plan-path ` (or `--plan-path -` to read the plan JSON from stdin). +- **MCP tool** — pass the plan inline as `plan` (or `plan_path`): + + ``` + flowx(command="route", parameters={"output_dir": "", "plan": { ...authored plan... }}) + ``` + +### The plan shape + +The plan carries **only** the decision (and an optional rationale) per component — and the decision is +the **customer's**, not yours. Your job is to present each component's members, its `recommended` +route, and both `options`, then serialize the customer's pick; you do not choose the route yourself. +The library recomputes `members`, `recommended`, and both `options` on record, so recorded facts +cannot drift from the inventory or be faked. Start from the recommendation's `default_plan` and flip +the components the customer chose to override: + +```json +{ + "components": [ + {"component_id": "component-1", "members": ["IngestSalesforce", "IngestWorkday"], "decision": "agentic", + "rationale": "Collapse both extractors onto one Lakeflow Connect pipeline"}, + {"component_id": "component-2", "members": ["BuildMart"], "decision": "deterministic"} + ] +} +``` + +Rules the validator enforces (all violations returned at once; nothing written on failure): + +- `decision` must be `"deterministic"` or `"agentic"`; `rationale` (optional) must be a non-empty + string when present. +- Every `member` must be a real inventory pipeline, and a component's `members` must **exactly match + one computed connected component** — a decision can never split a component or span two. +- The plan is a **bijection**: every component is decided exactly once (no partial plan, no + duplicate/conflicting decisions). +- `component_id` is **required** on every component — a non-empty string that must match the computed + component for those `members`. Start from the recommendation's `default_plan`, which already carries + the correct `component_id` for each component. + +### Triggering convert if the report is missing + +`route` edits `.work/translation_report.json`, which the convert phase produces. If it is missing, +pass `--source ` **and** `--source-path ` (MCP: `source` plus the source-specific +path param — `adf_source_path` or `airflow_source_path`; a generic `source_path` is **ignored**, and +ADF also accepts `adf_definitions` / `adf_volume_path` / `adf_workspace_path`) and `route` triggers +the convert phase in-process first. Otherwise run `flowx-convert` before routing. + +### venv CLI + +```bash +"$PY" -m flowx.adapter route --output-dir --plan-path plan.json \ + [--source adf --source-path ] # only needed to trigger convert when the report is missing +``` + +Exit 1 (nothing written) on a missing report it cannot produce, or a plan that fails validation. + +## Step 3 — Fill the routed-agentic groups + +Every routed-agentic pipeline's tasks are now `PlaceholderActivity` nodes with one pipeline-tagged +`AgenticGap` each. Two fills exist depending on the grain; you author the replacement (no LLM in the +library — it validates and merges). + +### 3a — Per-pipeline fill (keep the pipeline 1:1) + +When a routed-agentic pipeline stays one pipeline and you just author a Databricks task per gap, reuse +the **existing** name-matched merge — the same path the agentic gap flow uses. Write one JSON file per +resolved gap into an agentic-results directory: + +```json +{ + "activity_name": "", + "pipeline": "", + "task": {"type": "NotebookActivity", "name": "", "task_key": "", + "notebook_path": "/Workspace/.../your_translated_notebook"} +} +``` + +Then merge (`task_key`/`depends_on` are inherited from the matched task when omitted, preserving edges): + +```bash +"$PY" -m flowx.adapter convert --source adf --merge-agentic \ + --report /.work/translation_report.json \ + --agentic-results \ + [--output ] # default: overwrite --report +``` + +This per-pipeline merge is **ADF-only**. `--source adf` is **mandatory** — the convert phase runner +exits 2 without it, and `--source airflow` is explicitly rejected here (Airflow's `--merge-agentic` +is disabled and exits 2). For **Airflow** per-gap fills, use the **`flowx-resolve-airflow-gaps`** +skill (the fingerprint-bound `resolve-agentic` workflow), not this command. + +The merge finds, per result, the **first task whose `name` matches `activity_name`** (recursing into +`IfCondition`/`ForEach`/`Switch` containers) and replaces it — it matches by name and does **not** +verify the target is a `PlaceholderActivity`, so make sure each `activity_name` targets the intended +agentic gap. Other pipelines and unmatched tasks are left untouched. + +MCP: `flowx(command="merge_agentic", parameters={"source": "adf", "report_path": ..., "agentic_results_dir": ..., "output_path": ...})`. +The matched task is replaced in place by the authored task definition (carrying whatever fields that +task defines; the merge itself sets no `status`); the command exits non-zero if any result can't be +matched. + +### 3b — Cross-pipeline COMBINE (N pipelines → M) + +When the decision is a re-architecture that changes pipeline count — e.g. five extractor pipelines +collapse onto **one** Lakeflow Connect pipeline — use `fill-agentic combine`. You author the +replacement pipeline(s) as IR dicts and the whole routed group is swapped for them. + +```bash +"$PY" -m flowx.adapter fill-agentic combine \ + --output-dir \ + --members "IngestSalesforce,IngestWorkday" \ + --pipelines-path authored_pipelines.json \ + [--out ] +``` + +MCP: `flowx(command="fill_agentic", parameters={"output_dir": ..., "members": [...], "pipelines": [...]})` +(pass `pipelines` inline as a list, or `pipelines_path`). + +`combine` is the only action. Its guarantees: + +- `--members` (comma-separated; MCP accepts a list) must **exactly match** a routed-**agentic** + component in the recorded `metadata/conversion_plan.json`, whose `inventory_sha256` must still match + the current inventory. A partial group, a superset, a typo, or a deterministic component is refused + — you can't swap pipelines the plan didn't route agentic. +- `--pipelines-path` is a JSON **list** of pipeline IR dicts (the authored replacements), typically + carrying `AgenticComponentActivity` nodes (see below). +- Each authored pipeline **must** carry the source tag `"tags": {"source": "adf"}` (routing/agentic + conversion is ADF-only). Combine asserts this up front and **fails closed** (nothing written, with + a clear message) on a missing or non-`adf` tag, so a mis-tagged pipeline is caught here rather than + surviving to the package preflight. +- The merged report is **always** validated with the structural bundle invariants (a real + `prepare → write_bundle` pass) before it is written — no bypass — so a duplicate key, dangling + dependency, cycle, or dangling pipeline/run_job reference can never land on disk. On any violation, + `ok` is `false`, `violations` lists them, and nothing is written. + +### Authoring an `AgenticComponentActivity` (the escape hatch) + +When the target can't be expressed by the typed engine (e.g. a managed Lakeflow Connect ingestion +pipeline), emit an `AgenticComponentActivity` task inside the authored pipeline. The authored +**pipeline** carries the required `tags.source == "adf"`; the task carries the raw bundle components +the package phase writes verbatim: + +```json +{ + "name": "IngestAll", + "tags": {"source": "adf"}, + "tasks": [ + { + "name": "IngestAll", + "task_key": "ingest_all", + "type": "AgenticComponentActivity", + "files": [ + {"path": "src/ingest/lakeflow_connect.py", "content": "# authored pipeline source ..."} + ], + "resources": [ + {"resource_key": "ingest_all_pipeline", "definition": { "channel": "current", "...": "raw pipeline resource" }} + ], + "task": {"pipeline_task": {"pipeline_id": "${resources.pipelines.ingest_all_pipeline.id}"}}, + "raw_definition": { "...": "original source definitions, retained for auditing" } + } + ] +} +``` + +- `tags` — the authored pipeline **must** set `tags.source` to `"adf"` (routing/agentic is ADF-only); + combine fails closed otherwise (see the combine guarantees above). +- `files` — files written below the bundle `src/`; each is `path` + either UTF-8 `content` or + base64 `binary_content`. +- `resources` — bundle resources in the `resource_key` + raw `definition` shape the bundle writer + expects. +- `task` — the raw Databricks task fragment (`pipeline_task` or `notebook_task`) wiring to an + authored resource or file. +- `raw_definition` — the original source definition, retained for provenance. + +**Default the Lakeflow pipeline / Connect resource `channel` to `current` (stable/GA).** Do **not** +emit `channel: preview` by default. Let the enriched insight's **structured `release_state`** drive +the channel — the routing recommendation exposes each agentic option's states as a neutral +`release_disclosures` list (factual labels, no alarm). Surface the state as disclosure, tiered as: + +- `ga` — `channel: current`; **silent** (not surfaced). +- `unknown` — **silent**, treated exactly like `ga` (no separate label — we can't distinguish them). +- `public_preview` — the informational label **"Public Preview (production-ready)"**: generally + production-ready and supported per Databricks; still confirm workspace availability. Emit + `channel: preview` only with the cited `release_state_source`. +- `private_preview` — the plain factual label **"Private Preview"**. Emit `channel: preview` only + with the cited `release_state_source`. +- `beta` — the plain factual label **"Beta"**. Emit `channel: preview` only with the cited + `release_state_source`. + +Separately from *how the state is surfaced*, keep the recommendation-eligibility rule: treat a +**Private Preview** connector as `doNotSuggest` unless the workspace has confirmed +enrollment/entitlement. And **verify GA-vs-Preview status before recommending any connector or +Lakeflow Connect pattern**: do not hardcode GA/Preview status or dates (release state changes), check +the feature's current release state **and** target-workspace availability against the current public +Databricks docs, and cite the source. (Same structured grounding rule the `flowx-enrich` skill +applies when authoring recommended patterns.) + +## Step 4 — Continue to package + +Once the routed-agentic groups are filled and the report validates, continue with `flowx-convert`'s +just-in-time configuration (`inspect`/`modify`) as usual, then `flowx-package`. The recorded +`metadata/conversion_plan.json` is kept alongside `inventory.json` as the routing record. + +## Reference + +- `flowx-enrich` skill — the `insights` layer routing's agentic option consumes. +- `flowx-convert` — the deterministic engine, per-pipeline `merge_agentic`, and just-in-time config. +- `flowx-package` — turns the (filled) report into the deployable DAB bundle. diff --git a/src/flowx/adapter/__main__.py b/src/flowx/adapter/__main__.py index a2bb308..64ca32b 100644 --- a/src/flowx/adapter/__main__.py +++ b/src/flowx/adapter/__main__.py @@ -1,9 +1,9 @@ """Unified CLI entry point that the flowx skills and MCP tools drive via subprocesses. Exposes stateless subcommands -- the ``discover``/``convert``/``package`` phase runners plus -``inspect``, ``modify``, ``resolve-agentic``, ``enrich``, ``inputs``, ``materialize-lookup``, -``workspace-paths``, ``record-results``, and ``install-dashboard`` -- so each agent turn runs as an -independent process holding no session state across user prompts. +``inspect``, ``modify``, ``resolve-agentic``, ``enrich``, ``route``, ``fill-agentic``, ``inputs``, +``materialize-lookup``, ``workspace-paths``, ``record-results``, and ``install-dashboard`` -- so each +agent turn runs as an independent process holding no session state across user prompts. """ from __future__ import annotations @@ -87,6 +87,10 @@ def main(argv: list[str] | None = None) -> int: return _run_resolve_agentic(args) if args.command == "enrich": return _run_enrich(args) + if args.command == "route": + return _run_route(args) + if args.command == "fill-agentic": + return _run_fill_agentic(args) if args.command == "record-results": return _run_record_results(args) if args.command == "install-dashboard": @@ -163,6 +167,151 @@ def _run_enrich(args: argparse.Namespace) -> int: return 0 if result.get("ok") else 1 +def _run_route(args: argparse.Namespace) -> int: + """Implements the reshaped ``route``: one command that recommends, decides, records, and edits. + + Groups pipelines by connected component and computes the per-component recommendation (reusing + :mod:`flowx.routing`). Then: + + * with **no decision** on a non-TTY it emits the recommendation (components + both options + + findings + a ready-to-record default plan) and exits -- the dry run an agent reads first; + * with a **decision** -- ``--plan-path FILE`` (or ``--plan-path -`` to read the plan from stdin), + or an interactive TTY prompt -- it validates and records the fingerprint-bound + ``metadata/conversion_plan.json`` and then edits ``.work/translation_report.json`` + + ``gaps.json`` so every routed-agentic group's tasks become placeholder gaps (deterministic + groups untouched; nothing routed agentic leaves the report byte-identical). + + Triggers the convert phase in-process when the report is missing and ``--source`` / + ``--source-path`` are supplied. Returns 1 on a missing report it cannot produce, or on a plan + that fails validation (report + plan left untouched). + """ + from flowx import routing + from flowx.route_agentic import REPORT_FILENAME, WORK_DIRNAME, apply_plan_to_report + + inventory_path = args.output_dir / "metadata" / "inventory.json" + if not inventory_path.exists(): + print(f"No inventory.json under {inventory_path.parent}; run the discover phase first.", file=sys.stderr) + return 1 + try: + inventory = json.loads(inventory_path.read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError) as error: + print(f"Failed to read {inventory_path}: {error}", file=sys.stderr) + return 1 + recommendation = routing.build_recommendation(inventory) + + try: + plan = _route_decision(args, recommendation) + except (OSError, json.JSONDecodeError) as error: + print(f"Failed to read the conversion plan: {error}", file=sys.stderr) + return 1 + if plan is None: + # No decision on a non-TTY: emit the recommendation as a dry run and stop. + _emit_json(recommendation, args.out) + return 0 + + report_path = args.output_dir / WORK_DIRNAME / REPORT_FILENAME + if not report_path.exists(): + triggered = _trigger_convert(args) + if triggered != 0: + return triggered + if not report_path.exists(): + print( + f"No {REPORT_FILENAME} under {report_path.parent}; run the convert phase first " + "(or pass --source and --source-path so route can trigger it).", + file=sys.stderr, + ) + return 1 + + try: + result = routing.record_plan(args.output_dir, plan=plan) + except (FileNotFoundError, OSError, ValueError, json.JSONDecodeError) as error: + print(f"Failed to record conversion plan: {error}", file=sys.stderr) + return 1 + if not result.get("ok"): + _emit_json(result, args.out) + return 1 + try: + edit = apply_plan_to_report(args.output_dir, plan) + except (FileNotFoundError, OSError, ValueError, json.JSONDecodeError) as error: + print(f"Failed to edit translation report: {error}", file=sys.stderr) + return 1 + _emit_json({**result, "edit": edit}, args.out) + return 0 + + +def _route_decision(args: argparse.Namespace, recommendation: dict[str, Any]) -> dict[str, Any] | None: + """Resolve the routing decision as an authored plan, or ``None`` for a dry-run recommendation. + + Precedence: ``--plan-path FILE`` (a file), ``--plan-path -`` (stdin JSON), then -- when stdin is + a TTY -- an interactive per-component prompt. On a non-TTY with no ``--plan-path`` there is no + decision, so the caller emits the recommendation instead. + """ + from flowx.route_agentic import prompt_for_decisions + + if args.plan_path is not None: + if str(args.plan_path) == "-": + return json.loads(sys.stdin.read()) + return json.loads(Path(args.plan_path).read_text(encoding="utf-8")) + if sys.stdin.isatty(): + return prompt_for_decisions(recommendation) + return None + + +def _trigger_convert(args: argparse.Namespace) -> int: + """Run the convert phase in-process when ``--source`` / ``--source-path`` are supplied. + + Returns 0 when convert ran (or there was nothing to trigger because the inputs were absent), or + the convert phase's non-zero exit code on failure. + """ + source = getattr(args, "source", None) + source_path = getattr(args, "source_path", None) + if not source or not source_path: + return 0 + return _run_phase( + "convert", + ["--source", source, "--source-path", str(source_path), "--output-dir", str(args.output_dir)], + ) + + +def _run_fill_agentic(args: argparse.Namespace) -> int: + """Implements ``fill-agentic combine``: the cross-pipeline pipeline-grain fill. + + Replaces a routed-agentic group's pipelines (``--members`` as a comma-separated list) with the + agent-authored pipeline(s) read from ``--pipelines-path`` (a JSON list of pipeline IR dicts, each + typically carrying ``AgenticComponentActivity`` nodes). The membership must exactly match a + routed-agentic component in the recorded, fingerprint-bound ``metadata/conversion_plan.json``, and + the merged report is always validated structurally before it is written back -- there is no bypass. + Per-pipeline agentic fills reuse ``convert --merge-agentic`` instead and are not handled here. + """ + from flowx.route_agentic import apply_combine_fill + + if args.action != "combine": + print(f"Unknown fill-agentic action {args.action!r}; expected 'combine'.", file=sys.stderr) + return 2 + members = [member.strip() for member in args.members.split(",") if member.strip()] + if not members: + print("fill-agentic combine requires a non-empty --members list.", file=sys.stderr) + return 2 + try: + authored = json.loads(Path(args.pipelines_path).read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError) as error: + print(f"Failed to read {args.pipelines_path}: {error}", file=sys.stderr) + return 1 + if not isinstance(authored, list): + print("--pipelines-path must contain a JSON list of pipeline IR dicts.", file=sys.stderr) + return 2 + try: + result = apply_combine_fill(args.output_dir, members, authored) + except FileNotFoundError as error: + print(str(error), file=sys.stderr) + return 1 + except (OSError, ValueError, json.JSONDecodeError) as error: + print(f"Failed to apply combine fill: {error}", file=sys.stderr) + return 1 + _emit_json(result, args.out) + return 0 if result.get("ok") else 1 + + def _run_record_results(args: argparse.Namespace) -> int: """Implements ``record-results``: write per-pipeline coverage to a UC table. @@ -521,6 +670,83 @@ def _build_parser() -> argparse.ArgumentParser: help="Optional output file for the enrich result JSON; defaults to stdout.", ) + route = subparsers.add_parser( + "route", + help=( + "One command: recommend a per-connected-component route, take the decision (interactive on " + "a TTY, else --plan-path / stdin), record metadata/conversion_plan.json, and edit the " + "translation report so routed-agentic groups become placeholder gaps." + ), + ) + route.add_argument( + "--output-dir", + type=Path, + required=True, + help=( + "Migration output directory (reads metadata/inventory.json + .work/translation_report.json; " + "records metadata/conversion_plan.json and edits the report for routed-agentic groups)." + ), + ) + route.add_argument( + "--plan-path", + type=Path, + default=None, + help=( + "Path to the agent-authored conversion plan JSON to validate, record, and apply. Use '-' to " + "read the plan from stdin. Omit on a TTY for an interactive prompt, or on a non-TTY to emit " + "the recommendation as a dry run." + ), + ) + route.add_argument( + "--source", + default=None, + help="Migration source (adf | airflow); with --source-path, lets route trigger convert if needed.", + ) + route.add_argument( + "--source-path", + type=Path, + default=None, + help="Path to the source; with --source, lets route trigger the convert phase when the report is missing.", + ) + route.add_argument( + "--out", + type=Path, + default=None, + help="Optional output file for the recommendation / result JSON; defaults to stdout.", + ) + + fill_agentic = subparsers.add_parser( + "fill-agentic", + help=( + "Cross-pipeline COMBINE fill: replace a routed group's pipelines with agent-authored " + "pipeline(s), validated structurally before writing. Per-pipeline fills use convert --merge-agentic." + ), + ) + fill_agentic.add_argument("action", choices=("combine",), help="'combine' performs the pipeline-grain fill.") + fill_agentic.add_argument( + "--output-dir", + type=Path, + required=True, + help="Migration output directory (reads and rewrites .work/translation_report.json).", + ) + fill_agentic.add_argument( + "--members", + required=True, + help="Comma-separated pipeline names of the routed group to replace.", + ) + fill_agentic.add_argument( + "--pipelines-path", + type=Path, + required=True, + help="JSON file: a list of agent-authored pipeline IR dicts (typically AgenticComponentActivity nodes).", + ) + fill_agentic.add_argument( + "--out", + type=Path, + default=None, + help="Optional output file for the result JSON; defaults to stdout.", + ) + record = subparsers.add_parser( "record-results", help="Write per-pipeline migration coverage for this run to a Unity Catalog table.", diff --git a/src/flowx/bundler/dab_writer.py b/src/flowx/bundler/dab_writer.py index 93b8933..d71c7fe 100644 --- a/src/flowx/bundler/dab_writer.py +++ b/src/flowx/bundler/dab_writer.py @@ -378,6 +378,70 @@ def _default_report_path(output_dir: Path) -> Path: return work / "translation_report.json" +def _write_route_audit(output_dir: Path) -> Path | None: + """Persist a compact routing/gaps audit to ``metadata/route_audit.json`` before the prune. + + Package prunes the transient ``.work/`` folder (translation report + ``gaps.json``) by default, + which erases the "what did routing change?" trail. When a routing decision was recorded + (``metadata/conversion_plan.json`` exists), summarise the routed components, their decisions, and + the gaps routing introduced into an additive ``metadata/`` artifact that survives the prune. + + Returns the written path, or ``None`` when there is no recorded plan (no routing happened) -- so + the no-route path writes nothing and stays byte-identical. + """ + metadata_dir = Path(output_dir) / "metadata" + plan_path = metadata_dir / "conversion_plan.json" + if not plan_path.is_file(): + return None + try: + plan = json.loads(plan_path.read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError): + return None + if not isinstance(plan, dict): + return None + + components: list[dict[str, Any]] = [] + agentic_pipelines: list[str] = [] + for component in plan.get("components", []): + if not isinstance(component, dict): + continue + members = [str(member) for member in component.get("members", []) if isinstance(member, str)] + decision = component.get("decision") + components.append({"component_id": component.get("component_id"), "members": members, "decision": decision}) + if decision == "agentic": + agentic_pipelines.extend(members) + + gaps_introduced: list[dict[str, Any]] = [] + gaps_path = Path(output_dir) / ".work" / "gaps.json" + if gaps_path.is_file(): + try: + gaps = json.loads(gaps_path.read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError): + gaps = [] + for gap in gaps if isinstance(gaps, list) else []: + if isinstance(gap, dict): + gaps_introduced.append( + { + "pipeline": gap.get("pipeline"), + "activity_name": gap.get("activity_name"), + "activity_type": gap.get("activity_type"), + } + ) + + audit = { + "schema": "flowx.route_audit/v1", + "recorded_against_inventory_sha256": plan.get("inventory_sha256"), + "components": components, + "agentic_pipelines": sorted(set(agentic_pipelines)), + "gaps_count": len(gaps_introduced), + "gaps_introduced": gaps_introduced, + } + metadata_dir.mkdir(parents=True, exist_ok=True) + audit_path = metadata_dir / "route_audit.json" + audit_path.write_text(json.dumps(audit, indent=2), encoding="utf-8") + return audit_path + + def main(argv: list[str] | None = None) -> int: """Package-phase entry point for DAB bundle generation. @@ -584,6 +648,12 @@ def main(argv: list[str] | None = None) -> int: print(format_result(result), file=sys.stderr) invariant_violations += len(result.violations) + # Persist the routing/gaps audit to metadata/ before pruning .work/, so the "what did routing + # change?" trail survives the default prune. No-ops (writes nothing) when no plan was recorded. + audit_path = _write_route_audit(args.output_dir) + if audit_path is not None: + print(f"Wrote routing audit to {audit_path}") + if not args.keep_intermediates: work_dir = args.output_dir / ".work" if work_dir.is_dir(): diff --git a/src/flowx/mcp/server.py b/src/flowx/mcp/server.py index 0651640..c1bcbfa 100644 --- a/src/flowx/mcp/server.py +++ b/src/flowx/mcp/server.py @@ -53,6 +53,9 @@ def _transport_security() -> TransportSecuritySettings: flowx("discover", {"source": "adf", "adf_source_path": "...", "output_dir": "..."}) flowx("enrich", {"output_dir": "...", "insights": {...}}) # optional: merge agent-authored insights flowx("convert", {"source": "adf", "output_dir": "..."}) + flowx("route", {"output_dir": "..."}) # optional: recommend, then re-call with a plan + flowx("route", {"output_dir": "...", "plan": {...}}) # record the decision + edit the report + flowx("fill_agentic", {"output_dir": "...", "members": [...], "pipelines": [...]}) # combine fill (or merge_agentic) flowx("inspect", {"report_path": "/.work/translation_report.json"}) flowx("apply_answers", {"report_path": "...", "answers": ["id=value"], "output_dir": "..."}) flowx("package", {"output_dir": "...", "catalog": "main", "schema": "default"}) @@ -326,6 +329,100 @@ def _run(path: str) -> dict[str, Any]: return _run(str(inline_path)) +def _cmd_route(p: dict[str, Any]) -> dict[str, Any]: + """One command: recommend a per-component route, or record + apply the user's decision. + + With **no** ``plan`` / ``plan_path`` this emits the recommendation (components, both conversion + options per component, findings, and a ready-to-record default plan) -- the dry run an agent reads + first. With a decision (``plan`` inline or ``plan_path`` -- at most one) it validates and records + the fingerprint-bound ``metadata/conversion_plan.json`` AND edits ``.work/translation_report.json`` + + ``gaps.json`` so every routed-agentic group's tasks become placeholder gaps (deterministic groups + untouched). The returned ``ok`` reflects the CLI's success; ``result`` carries its JSON. + """ + output_dir = p.get("output_dir", "./flowx_output") + plan = p.get("plan") + plan_path = p.get("plan_path") + if plan is not None and plan_path is not None: + return {"ok": False, "error": "provide at most one of 'plan' (inline object) or 'plan_path'."} + + # Forward source + resolved source-path so the MCP path mirrors the CLI: route can then trigger + # convert when the report is missing. `source` is optional here (unlike discover/convert). + source_args, cleanup = _route_source_args(p) + try: + if plan is None and plan_path is None: + result = runner.run_adapter(["route", "--output-dir", output_dir, *source_args]) + return {"ok": result.ok, "result": runner.parse_stdout_json(result), "process": result.as_dict()} + + def _run(path: str) -> dict[str, Any]: + result = runner.run_adapter(["route", "--output-dir", output_dir, "--plan-path", path, *source_args]) + payload = runner.parse_stdout_json(result) + ok = bool(isinstance(payload, dict) and payload.get("ok")) + return {"ok": ok, "result": payload, "process": result.as_dict()} + + if plan_path is not None: + return _run(str(plan_path)) + with tempfile.TemporaryDirectory(prefix="flowx-plan-") as temporary: + inline_path = Path(temporary) / "conversion_plan.json" + inline_path.write_text(json.dumps(plan, indent=2), encoding="utf-8") + return _run(str(inline_path)) + finally: + cleanup() + + +def _route_source_args(p: dict[str, Any]) -> tuple[list[str], Callable[[], None]]: + """Resolve optional ``source`` / source-path into ``route`` CLI flags (empty when no source given). + + Mirrors :func:`_cmd_convert`'s source resolution so a materialized volume/workspace source is read + the same way, but ``source`` is optional for ``route`` -- absent it, no source flags are forwarded + and route simply skips the convert trigger. Returns ``(args, cleanup)``. + """ + if not p.get("source"): + return [], _noop + source_name = _source_name(p) + source, cleanup = _resolve_source(p) + args = ["--source", source_name] + if source: + args += ["--source-path", str(source)] + return args, cleanup + + +def _cmd_fill_agentic(p: dict[str, Any]) -> dict[str, Any]: + """Cross-pipeline COMBINE fill: replace a routed group's pipelines with agent-authored pipeline(s). + + ``members`` (a list of pipeline names, or a comma-separated string) names the routed group to + replace; the authored pipeline IR dicts come inline as ``pipelines`` (a JSON list) or via + ``pipelines_path`` (exactly one, staged to a temp file so the same CLI contract runs on both). The + merged report is validated structurally before it is written back; ``ok`` is False (nothing + written) when a structural invariant is violated. Per-pipeline agentic fills use ``merge_agentic``. + """ + output_dir = p.get("output_dir", "./flowx_output") + members = p.get("members") + if isinstance(members, list): + members = ",".join(str(member) for member in members) + if not members: + return {"ok": False, "error": "provide 'members' (a list of pipeline names or comma-separated string)."} + + pipelines = p.get("pipelines") + pipelines_path = p.get("pipelines_path") + if (pipelines is None) == (pipelines_path is None): + return {"ok": False, "error": "provide exactly one of 'pipelines' (inline list) or 'pipelines_path'."} + + def _run(path: str) -> dict[str, Any]: + args: list[Any] = ["fill-agentic", "combine", "--output-dir", output_dir, "--members", members] + args += ["--pipelines-path", path] + result = runner.run_adapter(args) + payload = runner.parse_stdout_json(result) + ok = bool(isinstance(payload, dict) and payload.get("ok")) + return {"ok": ok, "result": payload, "process": result.as_dict()} + + if pipelines_path is not None: + return _run(str(pipelines_path)) + with tempfile.TemporaryDirectory(prefix="flowx-fill-") as temporary: + inline_path = Path(temporary) / "authored_pipelines.json" + inline_path.write_text(json.dumps(pipelines, indent=2), encoding="utf-8") + return _run(str(inline_path)) + + def _cmd_inspect(p: dict[str, Any]) -> dict[str, Any]: args: list[Any] = ["inspect", p["report_path"]] for answer in p.get("answers") or []: @@ -529,6 +626,8 @@ def _cmd_install_dashboard(p: dict[str, Any]) -> dict[str, Any]: "merge_agentic": _cmd_merge_agentic, "resolve_agentic": _cmd_resolve_agentic, "enrich": _cmd_enrich, + "route": _cmd_route, + "fill_agentic": _cmd_fill_agentic, "inspect": _cmd_inspect, "apply_answers": _cmd_apply_answers, "materialize_lookup": _cmd_materialize_lookup, @@ -587,6 +686,23 @@ def flowx(command: str, parameters: dict[str, Any] | None = None) -> dict[str, A `insights` key (atomic, idempotent). `ok` reflects validation; `result.violations` lists any problems and the inventory is left untouched on failure. Author the insights by reading inventory.json + the source artifacts first (see the flowx-discover skill's insights guide). + - "route": output_dir(req), at most one of plan(inline object) | plan_path, plus optional + source + a source path (forwarded so route can trigger convert if the report is absent, like + the CLI) — one command that groups pipelines into connected components over control lineage + and routes each. With NO plan it emits the components with BOTH conversion options + (deterministic capability + motif/coverage evidence, and the agentic recommended patterns + with any simplification pattern surfaced), plus a ready-to-record default plan — the dry run + to read first. With a plan it validates + records metadata/conversion_plan.json AND edits the + translation report so every routed-agentic group's tasks become placeholder gaps + (deterministic groups untouched; convert's deterministic translation is never modified, and + with no agentic decision the report is byte-identical to today). + - "fill_agentic": output_dir(req), members(req: list of pipeline names or comma-separated + string), one of pipelines(inline list of pipeline IR dicts) | pipelines_path — cross-pipeline + COMBINE fill: replace a routed-agentic group's pipelines with the agent-authored pipeline(s) + (typically AgenticComponentActivity nodes). `members` must exactly match a routed-agentic + component in the recorded, fingerprint-bound conversion_plan.json, and the merged report is + always validated structurally before writing (no bypass). Per-pipeline agentic fills use + "merge_agentic" instead. - "inspect": report_path(req) — return the full translation-option schema (every option with a `show_when` condition) for the agent to walk locally. See "Collecting options" below. - "apply_answers": report_path(req), answers(req, list of "ID=VALUE"), output_dir, lookup_csv. diff --git a/src/flowx/models/conversion_plan.py b/src/flowx/models/conversion_plan.py new file mode 100644 index 0000000..9f56857 --- /dev/null +++ b/src/flowx/models/conversion_plan.py @@ -0,0 +1,157 @@ +"""Conversion-plan artifact (routing, #77) -- the user-approved per-component conversion decision. + +The routing step (:mod:`flowx.routing`) groups pipelines into connected components over control +lineage, presents each component's two conversion options as first-class peers, and records the +user's per-component choice as ``metadata/conversion_plan.json``. These models are **source-neutral** +and document the shape of that artifact; the validate/record engine works on the raw dict form and +these dataclasses back the unit tests, mirroring the split in :mod:`flowx.models.insights`. + +The agent authors **only** :attr:`ComponentPlan.decision` (and an optional +:attr:`ComponentPlan.rationale`). Everything else -- ``component_id``, ``members``, ``recommended``, +and both :class:`ComponentOptions` -- is recomputed by the library on record so the recorded facts +can never drift from the inventory or be faked. The library also owns :attr:`ConversionPlan.schema_version` +and :attr:`ConversionPlan.inventory_sha256` (the fingerprint that binds the plan to the inventory). + +This is Phase-1, descriptive-only routing metadata: recording a plan does not alter ``convert``. +""" + +from __future__ import annotations + +from dataclasses import dataclass, field +from typing import Any, Literal + +# The plan schema version stamped onto the recorded artifact. Bump on any backwards-incompatible +# change to the recorded shape. +SCHEMA_VERSION = "1" + +# The two conversion routes a component can take. +DECISION_DETERMINISTIC = "deterministic" +DECISION_AGENTIC = "agentic" +DECISIONS: tuple[str, ...] = (DECISION_DETERMINISTIC, DECISION_AGENTIC) + +# Recommended-pattern release states surfaced as a neutral disclosure label on the agentic option +# (:attr:`AgenticOption.release_disclosures`). ``"ga"`` and ``"unknown"`` are deliberately **silent** +# -- they contribute no entry, and ``"unknown"`` is treated exactly like ``"ga"`` (we do not surface +# or distinguish it). This is factual labelling, never a warning or an alarm. +DISCLOSED_RELEASE_STATES: tuple[str, ...] = ("public_preview", "private_preview", "beta") + + +@dataclass(slots=True, kw_only=True) +class DeterministicOption: + """The deterministic (1:1 engine) conversion option for a component. + + Attributes: + capable: ``True`` when every activity in the component is engine-capable -- each either has a + ``"deterministic"`` strategy or is claimed by a detected motif -- so the whole component + can convert deterministically with no gap. + activity_counts: Count of activities by strategy bucket (``deterministic`` / ``agentic`` / + ``unsupported``) across the component's pipelines. + motifs: The motif ids detected in the component (the #64 multi-activity capability signal), + sorted and de-duplicated. + uncovered: One entry per activity that keeps the component from being fully deterministic -- + a dict of ``pipeline`` / ``activity`` / ``type`` / ``strategy``. Empty when ``capable``. + """ + + capable: bool + activity_counts: dict[str, int] = field(default_factory=dict) + motifs: list[str] = field(default_factory=list) + uncovered: list[dict[str, Any]] = field(default_factory=list) + + +@dataclass(slots=True, kw_only=True) +class AgenticPattern: + """One recommended Databricks pattern for the agentic option, drawn from a pipeline's insights. + + Attributes: + pipeline: The member pipeline the pattern was recommended for. + pattern: The named, publicly-documented Databricks capability (verbatim from the insight). + fit: One line on why it fits / what it replaces. + simplification_pattern: ``True`` when the pattern uses a distinctive capability that collapses + a whole legacy pattern (e.g. a multi-pipeline -> Lakeflow Connect re-architecture). + release_state: The pattern's verified Databricks GA/Preview release state, carried verbatim + from the insight (one of :data:`~flowx.models.insights.RELEASE_STATES`, or ``None`` when + the insight left it unstated). Drives the neutral release-state disclosure on + :class:`AgenticOption`. + release_state_source: The doc URL / citation grounding :attr:`release_state`, carried verbatim + from the insight; ``None`` when unstated. + """ + + pipeline: str + pattern: str + fit: str + simplification_pattern: bool + release_state: Literal["ga", "public_preview", "private_preview", "beta", "unknown"] | None = None + release_state_source: str | None = None + + +@dataclass(slots=True, kw_only=True) +class AgenticOption: + """The agentic conversion option for a component. + + Attributes: + recommended_patterns: The recommended patterns gathered from the member pipelines' insights, + each tagged with its pipeline. Empty when the inventory carries no insights. + has_simplification: ``True`` when any recommended pattern is a ``simplification_pattern`` -- + surfaced prominently so the user sees a re-architecture option, not a buried sub-key. + release_disclosures: A neutral, per-pattern **disclosure** of any non-silent ``release_state`` + -- one entry per recommended pattern whose state is in :data:`DISCLOSED_RELEASE_STATES`, + carrying the ``pipeline``, ``pattern``, ``release_state``, and a factual ``label`` + (``public_preview`` labelled "Public Preview (production-ready)"; ``private_preview`` / + ``beta`` stated as the plain labels "Private Preview" / "Beta"). ``"ga"`` and ``"unknown"`` + are **silent** -- they add nothing (``"unknown"`` is treated exactly like ``"ga"``). Empty + when nothing needs disclosing. Factual labelling, never a warning or an alarm. + """ + + recommended_patterns: list[AgenticPattern] = field(default_factory=list) + has_simplification: bool = False + release_disclosures: list[dict[str, Any]] = field(default_factory=list) + + +@dataclass(slots=True, kw_only=True) +class ComponentOptions: + """Both conversion options for a component, as first-class peers.""" + + deterministic: DeterministicOption + agentic: AgenticOption + + +@dataclass(slots=True, kw_only=True) +class ComponentPlan: + """One component's routing decision. + + Attributes: + component_id: Stable id assigned by the library (``"component-"``). + members: The component's pipeline names, sorted (library-computed). + recommended: The library's starting suggestion -- ``"deterministic"`` when the component is + engine-capable, else ``"agentic"``. + decision: The user's authored per-component choice (may override :attr:`recommended`). + options: Both conversion options with their evidence (library-computed). Optional here so the + authored input -- which carries only the decision -- can round-trip through this model. + rationale: Optional author note on why this decision was chosen. + """ + + component_id: str + members: list[str] = field(default_factory=list) + recommended: str | None = None + decision: str + options: ComponentOptions | None = None + rationale: str | None = None + + +@dataclass(slots=True, kw_only=True) +class ConversionPlan: + """The recorded conversion-plan artifact. + + Attributes: + components: One :class:`ComponentPlan` per connected component. + findings: Human-readable notes about unresolved/dangling control edges retained during + component computation (never silently severed). + schema_version: Library-owned plan schema version. + inventory_sha256: Library-owned fingerprint binding the plan to the deterministic inventory + base (see :func:`flowx.discovery_insights.inventory_fingerprint`). + """ + + components: list[ComponentPlan] = field(default_factory=list) + findings: list[str] = field(default_factory=list) + schema_version: str = SCHEMA_VERSION + inventory_sha256: str | None = None diff --git a/src/flowx/route_agentic.py b/src/flowx/route_agentic.py new file mode 100644 index 0000000..d21f30b --- /dev/null +++ b/src/flowx/route_agentic.py @@ -0,0 +1,541 @@ +"""Phase-1 in-engine agentic conversion: apply a routing decision, then fill the gaps. + +``convert`` always produces a deterministic ``.work/translation_report.json`` -- that translation is +**unchanged** by this module. Routing then decides, per connected component, whether each part +converts deterministically or agentically (:mod:`flowx.routing` computes the recommendation and +records the fingerprint-bound ``metadata/conversion_plan.json``). This module carries out the +**post-convert alteration and fill** the decision implies, always keeping the work in-engine so the +package phase, structural validation, and provenance all still apply: + +* :func:`alter_report` rewrites the report for the routed-**agentic** groups only: every task in an + agentic-routed pipeline is removed and replaced by a :class:`~flowx.models.ir.PlaceholderActivity`, + and exactly one pipeline-tagged :class:`~flowx.models.ir.AgenticGap` is emitted per routed task, + replacing (never appending to) any prior gap for those pipelines' tasks, so the standard gap-fill + path handles them without duplicates and the edit is idempotent. Pipelines in deterministic groups + are left byte-identical, and when nothing is routed agentic the report and gaps are returned + unchanged -- the non-breaking guarantee. +* the agent authors the fill. **Per-pipeline** agentic reuses the existing name-matched + :func:`flowx.ir_serde.merge_agentic_results` (no new code here). **Cross-pipeline COMBINE** + (N pipelines -> M, e.g. one Lakeflow Connect pipeline) is the one net-new capability: + :func:`combine_group_fill` swaps the routed group's pipelines for the agent-authored pipeline(s), + which carry :class:`~flowx.models.ir.AgenticComponentActivity` nodes (the escape hatch). +* :func:`validate_report_structurally` packages the merged report through the same + ``prepare -> write_bundle`` path the package phase uses and runs the existing + :func:`flowx.validate.bundle_invariants.check_bundle_dir` over the output, so a fill can never + introduce duplicate keys, dangling dependencies, a cycle, or a dangling pipeline/run_job reference. + +There is no LLM here: the tool edits the report and validates the authored fill; the fill itself is +supplied by the agent/harness, following the same ask -> author -> continue pattern as the existing +agentic flow. +""" + +from __future__ import annotations + +import copy +import json +import os +import sys +import tempfile +from collections.abc import Callable, Iterable +from pathlib import Path +from typing import Any + +from flowx.models.conversion_plan import DECISION_AGENTIC + +# The report + gaps live under the shared output dir's transient .work/ folder, beside the pipeline IR. +WORK_DIRNAME = ".work" +REPORT_FILENAME = "translation_report.json" +GAPS_FILENAME = "gaps.json" +# The recorded plan + inventory the combine fill binds against live under metadata/. +METADATA_DIRNAME = "metadata" +INVENTORY_FILENAME = "inventory.json" + +# Routing / in-engine agentic conversion is ADF-only, so every agent-authored combine pipeline must +# carry this source tag. The package preflight enforces it too, but combine asserts it up front so a +# mis-tagged authored pipeline fails closed here (nothing written) instead of surviving to package. +REQUIRED_COMBINE_SOURCE_TAG = "adf" + +# Additive top-level marker stamped onto the report each time a combine is applied, recording the +# (component_id, inventory fingerprint) of every applied combine. Idempotency keys off this recorded +# state -- not off comparing authored pipeline names to member names -- so a re-run is detected no +# matter what the authored replacement is named or whether a name collides with a former member. It +# lives *in* the report, so a fresh convert (which rewrites the report) naturally clears it, and it is +# ignored by the package phase (not one of the recognized report shape keys) and by ir_serde (which +# reads per-pipeline IR, not the report wrapper). +_COMBINE_PROVENANCE_KEY = "_combine_provenance" + +# Guidance stamped onto every placeholder the alteration produces. +_PLACEHOLDER_COMMENT = ( + "Routed agentic by the conversion plan; author a replacement task (per-pipeline fill) or replace " + "the whole group with agent-authored pipeline(s) (cross-pipeline combine)." +) + + +# --------------------------------------------------------------------------- # +# Reading the recorded / authored plan. +# --------------------------------------------------------------------------- # + + +def agentic_pipeline_names(plan: dict[str, Any]) -> set[str]: + """Return the member pipelines of every component the plan decides ``"agentic"``. + + Reads either an authored plan (the agent-supplied ``{"components": [...]}``) or the recorded + ``conversion_plan.json`` -- both carry ``members`` and ``decision`` per component. + """ + names: set[str] = set() + for component in plan.get("components", []): + if isinstance(component, dict) and component.get("decision") == DECISION_AGENTIC: + names.update(str(member) for member in component.get("members", []) if isinstance(member, str)) + return names + + +# --------------------------------------------------------------------------- # +# Interactive decision prompt (one route from the user's seat). +# --------------------------------------------------------------------------- # + + +def _stderr(line: str) -> None: + """Default sink for the interactive prompt's narration -- stderr keeps stdout clean for JSON.""" + print(line, file=sys.stderr) + + +def prompt_for_decisions( + recommendation: dict[str, Any], + *, + input_fn: Callable[[str], str] | None = None, + output_fn: Callable[[str], None] | None = None, +) -> dict[str, Any]: + """Ask the user, per connected component, whether to convert deterministically or agentically. + + Shows each component's members, the library's recommendation, and -- when the agentic option + carries a re-architecture (e.g. a multi-pipeline -> Lakeflow Connect simplification) -- flags it + prominently. An empty answer accepts the recommendation; ``d``/``a`` (or the full words) choose + a route. Returns an authored plan dict (``{"components": [{component_id, members, decision}]}``) + ready to hand to :func:`flowx.routing.record_plan`. + + ``input_fn`` and ``output_fn`` are injectable so the prompt is exercised without a real TTY; both + are resolved at call time (defaulting to the builtin ``input`` and a stderr sink) so a test that + patches ``builtins.input`` is honoured. + """ + ask = input_fn if input_fn is not None else input + say = output_fn if output_fn is not None else _stderr + components_out: list[dict[str, Any]] = [] + for component in recommendation.get("components", []): + component_id = str(component.get("component_id")) + members = list(component.get("members", [])) + recommended = str(component.get("recommended", "deterministic")) + options = component.get("options") or {} + has_simplification = bool((options.get("agentic") or {}).get("has_simplification")) + + say(f"\nComponent {component_id}: {', '.join(members) or '(none)'}") + say(f" recommended: {recommended}") + if has_simplification: + say(" agentic option includes a simplification re-architecture (e.g. Lakeflow Connect)") + answer = ask(f" route [d]eterministic / [a]gentic (default={recommended}): ").strip().lower() + if answer in ("a", "agentic"): + decision = DECISION_AGENTIC + elif answer in ("d", "deterministic"): + decision = "deterministic" + else: + decision = recommended + components_out.append({"component_id": component_id, "members": members, "decision": decision}) + return {"components": components_out} + + +# --------------------------------------------------------------------------- # +# The edit step: rewrite the report for routed-agentic groups. +# --------------------------------------------------------------------------- # + + +def _report_pipelines(report: dict[str, Any]) -> list[dict[str, Any]]: + """The pipeline dicts in a report, whether it is single-pipeline or a ``{"pipelines": [...]}`` wrapper.""" + if isinstance(report, dict) and "pipelines" in report and isinstance(report["pipelines"], list): + return [pipeline for pipeline in report["pipelines"] if isinstance(pipeline, dict)] + return [report] + + +def _placeholder_and_gap(task: dict[str, Any], pipeline_name: str) -> tuple[dict[str, Any], dict[str, Any]]: + """Turn one task into a placeholder + gap, preserving its identity and edges. + + Idempotent: a task that is already a ``PlaceholderActivity`` (a re-run of the edit) is kept as-is + and its gap is rebuilt from the recorded ``original_type`` / ``raw_definition`` rather than + wrapping the placeholder in another placeholder. + """ + name = task.get("name") + if task.get("type") == "PlaceholderActivity": + original_type = str(task.get("original_type", "unknown")) + gap: dict[str, Any] = { + "activity_name": name, + "activity_type": original_type, + "raw_definition": task.get("raw_definition"), + "pipeline": pipeline_name, + } + return task, gap + + original_type = str(task.get("type", "unknown")) + placeholder: dict[str, Any] = { + "name": name, + "task_key": task.get("task_key"), + "type": "PlaceholderActivity", + "original_type": original_type, + "comment": _PLACEHOLDER_COMMENT, + # Keep the deterministic translation as context so the agent can author against it. + "raw_definition": task, + } + if task.get("depends_on"): + placeholder["depends_on"] = task["depends_on"] + gap = { + "activity_name": name, + "activity_type": original_type, + "raw_definition": task, + "pipeline": pipeline_name, + } + return placeholder, gap + + +def alter_report( + report: dict[str, Any], + gaps: list[dict[str, Any]], + agentic_pipelines: Iterable[str], +) -> tuple[dict[str, Any], list[dict[str, Any]]]: + """Rewrite the report so every routed-agentic pipeline's tasks become placeholder gaps. + + Returns ``(new_report, new_gaps)``. Pipelines not in ``agentic_pipelines`` are copied through + untouched. When ``agentic_pipelines`` is empty the inputs are returned unchanged (identity), so a + no-decision / all-deterministic route leaves today's output byte-for-byte intact. + """ + agentic = {name for name in agentic_pipelines} + if not agentic: + return report, gaps + + report = copy.deepcopy(report) + + # Task names owned by pipelines that are NOT routed agentic. Convert gaps are untagged (they carry + # no pipeline), so a name that also belongs to a non-routed pipeline is ambiguous: dropping it + # could remove that pipeline's gap, so it is preserved. Computed before mutating routed pipelines. + nonrouted_task_names: set[str] = set() + for pipeline in _report_pipelines(report): + if pipeline.get("name") in agentic: + continue + for task in pipeline.get("tasks", []): + if isinstance(task, dict): + nonrouted_task_names.add(str(task.get("name"))) + + # Emit exactly one pipeline-tagged gap per routed task, replacing (never appending to) any prior + # gap for a routed pipeline's tasks. Without this, a task that convert already recorded as an + # agentic gap -- or a re-run of the edit -- would leave duplicate/again-appended gaps. + fresh_gaps: list[dict[str, Any]] = [] + routed_task_names: set[str] = set() + for pipeline in _report_pipelines(report): + if pipeline.get("name") not in agentic: + continue + placeholders: list[dict[str, Any]] = [] + for task in pipeline.get("tasks", []): + if not isinstance(task, dict): + continue + placeholder, gap = _placeholder_and_gap(task, str(pipeline.get("name"))) + placeholders.append(placeholder) + fresh_gaps.append(gap) + routed_task_names.add(str(task.get("name"))) + pipeline["tasks"] = placeholders + + kept_gaps: list[dict[str, Any]] = [] + for gap in gaps: + pipeline_name = gap.get("pipeline") if isinstance(gap, dict) else None + activity_name = gap.get("activity_name") if isinstance(gap, dict) else None + # Drop our own prior tagged gaps for now-routed pipelines (idempotent re-run). + if pipeline_name in agentic: + continue + # Drop an untagged convert gap only when the name belongs *exclusively* to a routed pipeline -- + # a fresh tagged gap now supersedes it. When the same name also names a task in a non-routed + # pipeline, the gap is ambiguous and kept, so a non-routed pipeline never loses its gap. + if pipeline_name is None and activity_name in routed_task_names and activity_name not in nonrouted_task_names: + continue + kept_gaps.append(gap) + return report, kept_gaps + fresh_gaps + + +def _write_json_atomic(path: Path, document: Any) -> None: + """Write JSON atomically (temp file + ``os.replace``), matching the report's 2-space indentation.""" + path.parent.mkdir(parents=True, exist_ok=True) + temporary = path.with_name(f".{path.name}.tmp") + temporary.write_text(json.dumps(document, indent=2, default=str), encoding="utf-8") + os.replace(temporary, path) + + +def apply_plan_to_report(output_dir: Path, plan: dict[str, Any]) -> dict[str, Any]: + """Apply a routing decision to the report on disk: placeholder the routed-agentic groups. + + Reads ``/.work/translation_report.json`` (and ``gaps.json`` when present), rewrites + them for the plan's agentic components, and writes them back atomically. When no component is + routed agentic the files are left untouched, so the non-breaking guarantee holds. + + Returns a summary dict with the altered pipeline names and the resulting gap count. + + Raises: + FileNotFoundError: when the translation report is missing (run convert first). + """ + work = Path(output_dir) / WORK_DIRNAME + report_path = work / REPORT_FILENAME + if not report_path.exists(): + raise FileNotFoundError(f"No {REPORT_FILENAME} under {work}; run the convert phase first.") + gaps_path = work / GAPS_FILENAME + + agentic = agentic_pipeline_names(plan) + if not agentic: + return {"agentic_pipelines": [], "gaps": 0, "altered": False} + + report = json.loads(report_path.read_text(encoding="utf-8")) + gaps = json.loads(gaps_path.read_text(encoding="utf-8")) if gaps_path.exists() else [] + if not isinstance(gaps, list): + gaps = [] + + new_report, new_gaps = alter_report(report, gaps, agentic) + _write_json_atomic(report_path, new_report) + _write_json_atomic(gaps_path, new_gaps) + return {"agentic_pipelines": sorted(agentic), "gaps": len(new_gaps), "altered": True} + + +# --------------------------------------------------------------------------- # +# Cross-pipeline COMBINE: the pipeline-grain fill. +# --------------------------------------------------------------------------- # + + +def combine_group_fill( + report: dict[str, Any], + group_members: Iterable[str], + authored_pipelines: list[dict[str, Any]], +) -> dict[str, Any]: + """Replace a routed group's pipelines with the agent-authored pipeline(s). + + Drops every pipeline whose name is in ``group_members`` and appends the ``authored_pipelines`` + (each a pipeline IR dict, typically carrying ``AgenticComponentActivity`` nodes). Pipelines + outside the group are preserved in order. Returns a ``{"pipelines": [...]}`` report; the input is + not mutated. + """ + members = {name for name in group_members} + kept = [copy.deepcopy(pipeline) for pipeline in _report_pipelines(report) if pipeline.get("name") not in members] + kept.extend(copy.deepcopy(pipeline) for pipeline in authored_pipelines) + return {"pipelines": kept} + + +def _resolve_agentic_component(output_dir: Path, members: set[str]) -> tuple[str | None, str | None, str | None]: + """Bind ``members`` to a routed-**agentic** component in the recorded, fingerprint-bound plan. + + Reads ``metadata/conversion_plan.json`` and ``metadata/inventory.json`` and returns + ``(component_id, inventory_fingerprint, error)``. The error is set (and the other two are ``None``) + when: the plan or inventory is missing; the plan's ``inventory_sha256`` no longer matches the + current inventory (a stale plan); ``members`` do not exactly equal one component's members (a + partial, superset, or mistyped group); or the exactly-matching component is routed deterministic + rather than agentic. Requiring an exact match to a routed-agentic component stops a caller from + swapping deterministic pipelines or a partial group. The fingerprint is returned so the combine can + record its provenance keyed on the exact inventory it was applied against. + """ + from flowx.discovery_insights import inventory_fingerprint + + metadata = Path(output_dir) / METADATA_DIRNAME + plan_path = metadata / "conversion_plan.json" + inventory_path = metadata / INVENTORY_FILENAME + if not plan_path.exists(): + return None, None, "No metadata/conversion_plan.json; record a routing decision with `route` first." + if not inventory_path.exists(): + return None, None, "No metadata/inventory.json; run the discover phase first." + + plan = json.loads(plan_path.read_text(encoding="utf-8")) + inventory = json.loads(inventory_path.read_text(encoding="utf-8")) + recorded_fingerprint = plan.get("inventory_sha256") + current_fingerprint = inventory_fingerprint(inventory) + if recorded_fingerprint != current_fingerprint: + return ( + None, + None, + ( + "conversion_plan.json is stale: it was recorded against a different inventory " + f"({recorded_fingerprint!r} != {current_fingerprint!r}); re-run `route` before filling." + ), + ) + + for component in plan.get("components", []): + if not isinstance(component, dict): + continue + component_members = {str(member) for member in component.get("members", [])} + if component_members != members: + continue + if component.get("decision") == DECISION_AGENTIC: + return str(component.get("component_id")), current_fingerprint, None + return ( + None, + None, + ( + f"members {sorted(members)} match component {component.get('component_id')!r}, which is " + f"routed {component.get('decision')!r}, not agentic; only routed-agentic components can be combined." + ), + ) + return ( + None, + None, + ( + f"members {sorted(members)} do not exactly match any component in the recorded plan " + "(partial, superset, or mistyped); pass the exact member set of one routed-agentic component." + ), + ) + + +def _authored_source_tag_violations(authored_pipelines: list[dict[str, Any]]) -> list[str]: + """Reports each authored combine pipeline that is missing the required ``tags.source == 'adf'``. + + Routing / in-engine agentic conversion is ADF-only, so an authored combine pipeline that omits or + mis-sets the source tag is an authoring error. Catching it here fails the combine closed (nothing + written) with a clear message, rather than letting the mis-tagged pipeline reach the package + preflight where it is only rejected much later. + """ + violations: list[str] = [] + for index, pipeline in enumerate(authored_pipelines): + label = pipeline.get("name") if isinstance(pipeline, dict) and pipeline.get("name") else f"pipeline[{index}]" + tags = pipeline.get("tags") if isinstance(pipeline, dict) else None + source = tags.get("source") if isinstance(tags, dict) else None + if source != REQUIRED_COMBINE_SOURCE_TAG: + violations.append( + f"{label}: authored combine pipeline must carry tags.source == " + f"{REQUIRED_COMBINE_SOURCE_TAG!r}, got {source!r}" + ) + return violations + + +def _combine_already_applied(report: dict[str, Any], component_id: str, fingerprint: str) -> bool: + """Report whether this component's combine (at this inventory fingerprint) is already recorded. + + Reads the additive ``_combine_provenance`` marker stamped onto the report by a prior combine. + Detection is keyed purely on ``(component_id, fingerprint)`` -- never on comparing authored + pipeline names to member names -- so a re-run is recognised as already-combined no matter what the + authored replacement is named, and even when an authored name collides with a former member. + """ + provenance = report.get(_COMBINE_PROVENANCE_KEY) if isinstance(report, dict) else None + if not isinstance(provenance, list): + return False + return any( + isinstance(entry, dict) + and entry.get("component_id") == component_id + and entry.get("inventory_sha256") == fingerprint + for entry in provenance + ) + + +def apply_combine_fill( + output_dir: Path, + group_members: Iterable[str], + authored_pipelines: list[dict[str, Any]], +) -> dict[str, Any]: + """Combine a routed-agentic group into agent-authored pipeline(s) on disk. + + The group's membership is **bound to the recorded plan**: ``group_members`` must exactly match a + routed-agentic component in ``metadata/conversion_plan.json`` (whose fingerprint must still match + the current inventory), so a caller cannot swap deterministic pipelines or a partial/typoed group. + Every authored pipeline must carry ``tags.source == 'adf'`` (routing/agentic is ADF-only); a + mis-tagged pipeline fails the combine closed here rather than surviving to the package preflight. + The merged report is then **always** validated with the structural bundle invariants (a real + ``prepare -> write_bundle`` pass over :func:`validate_report_structurally`) -- there is no bypass -- + and written back only when it passes, so a dangling reference or duplicate key never lands on disk. + + The combine is **idempotent**, keyed on recorded report state rather than pipeline names. Each + successful combine stamps a ``_combine_provenance`` entry -- ``(component_id, inventory + fingerprint)`` -- onto the report. A re-run detects that stamp and no-ops (``already_combined`` + true) instead of collapsing/appending again, so it never duplicates the authored pipeline(s) even + when the authored replacement is renamed, and it still recognises the already-combined state when + an authored pipeline's name collides with a former member. A fresh ``convert`` rewrites the report + without the marker, so the combine will re-apply after a genuine re-convert. + + Returns ``{"ok", "violations", "error", "component_id", "pipelines", "already_combined"}``. ``ok`` + is ``False`` (and nothing written) on a plan/membership error (``error`` set), a missing source tag, + or any structural violation (``violations`` set). + + Raises: + FileNotFoundError: when the translation report is missing (run convert first). + """ + members = {str(member) for member in group_members} + component_id, fingerprint, error = _resolve_agentic_component(output_dir, members) + if error is not None: + return {"ok": False, "error": error, "violations": [], "pipelines": 0} + assert component_id is not None and fingerprint is not None # guaranteed when error is None + + tag_violations = _authored_source_tag_violations(authored_pipelines) + if tag_violations: + return {"ok": False, "error": None, "violations": tag_violations, "pipelines": 0} + + work = Path(output_dir) / WORK_DIRNAME + report_path = work / REPORT_FILENAME + if not report_path.exists(): + raise FileNotFoundError(f"No {REPORT_FILENAME} under {work}; run the convert phase first.") + + report = json.loads(report_path.read_text(encoding="utf-8")) + + # Idempotency: a prior combine stamps _combine_provenance onto the report for this + # (component_id, fingerprint). Detecting that recorded state -- not the authored/member name sets + # -- means a re-run no-ops regardless of how the authored replacement is named or whether a name + # collides with a former member, so it can never append a duplicate authored pipeline. + if _combine_already_applied(report, component_id, fingerprint): + return { + "ok": True, + "error": None, + "violations": [], + "component_id": component_id, + "pipelines": len(_report_pipelines(report)), + "already_combined": True, + } + + merged = combine_group_fill(report, members, authored_pipelines) + # Carry any prior provenance forward (combine_group_fill returns only ``pipelines``) and record + # this combine so a later re-run detects it. + prior_provenance = report.get(_COMBINE_PROVENANCE_KEY) if isinstance(report, dict) else None + provenance = ( + [entry for entry in prior_provenance if isinstance(entry, dict)] if isinstance(prior_provenance, list) else [] + ) + provenance.append({"component_id": component_id, "inventory_sha256": fingerprint, "members": sorted(members)}) + merged[_COMBINE_PROVENANCE_KEY] = provenance + + result = validate_report_structurally(merged) + if not result.ok: + violations = [f"[{finding.code}] {finding.location}: {finding.message}" for finding in result.violations] + return {"ok": False, "error": None, "violations": violations, "pipelines": 0} + + _write_json_atomic(report_path, merged) + return { + "ok": True, + "error": None, + "violations": [], + "component_id": component_id, + "pipelines": len(merged["pipelines"]), + "already_combined": False, + } + + +# --------------------------------------------------------------------------- # +# Structural validation via the existing bundle invariants. +# --------------------------------------------------------------------------- # + + +def validate_report_structurally(report: dict[str, Any]) -> Any: + """Package the report to a throwaway bundle and run the existing structural invariants over it. + + Uses the same ``prepare_workflow -> write_bundle`` path as the package phase, then aggregates + :func:`flowx.validate.bundle_invariants.check_bundle_dir` findings across every emitted bundle so + duplicate keys, dangling dependencies, cycles, and dangling pipeline/run_job references are all + caught before the merged report is trusted. Returns a + :class:`~flowx.validate.bundle_invariants.BundleInvariantResult`. + """ + from flowx.bundler.dab_writer import pipeline_dict_to_ir, write_bundle + from flowx.preparer.workflow_preparer import prepare_workflow + from flowx.utils import normalize_task_key + from flowx.validate.bundle_invariants import BundleInvariantResult, check_bundle_dir + + pipelines = _report_pipelines(report) + findings: list[Any] = [] + with tempfile.TemporaryDirectory(prefix="flowx-fill-validate-") as temporary: + root = Path(temporary) + for pipeline_dict in pipelines: + pipeline, _skipped = pipeline_dict_to_ir(pipeline_dict) + workflow = prepare_workflow(pipeline) + bundle_dir = root / normalize_task_key(pipeline.name) if len(pipelines) > 1 else root + write_bundle(workflow, bundle_dir) + findings.extend(check_bundle_dir(bundle_dir).findings) + return BundleInvariantResult(findings=findings) diff --git a/src/flowx/routing.py b/src/flowx/routing.py new file mode 100644 index 0000000..be2ce9d --- /dev/null +++ b/src/flowx/routing.py @@ -0,0 +1,591 @@ +"""Compute a per-connected-component conversion route over the discover inventory (#77). + +The discover phase writes a deterministic ``metadata/inventory.json``; :mod:`flowx.discovery_insights` +optionally enriches it with an additive ``insights`` block. This module turns those signals into a +**routing recommendation** and records the **user's decision** as a fingerprint-bound +``metadata/conversion_plan.json`` artifact -- a Phase-1, descriptive-only step: + +* pipelines are grouped into weak/undirected **connected components** over the inventory's control + lineage (``lineage.control_edges``), so mutually-referencing pipelines are decided together and a + caller/callee reference is never split across incompatible routes; +* each component surfaces **both conversion options as first-class peers** -- a *deterministic* + option (the engine-capability assessment: every activity engine-capable via its ``strategy`` or + claimed by a detected motif, plus its motif/coverage evidence and any uncovered gaps) and an + *agentic* option (the recommended Databricks patterns from the ``insights`` block, with any + ``simplification_pattern`` such as a multi-pipeline -> Lakeflow Connect re-architecture surfaced + prominently) -- so the user can choose per group; +* ``recommended`` is a library-computed starting suggestion (deterministic when the whole component + is engine-capable), and ``decision`` is the user's authored per-component choice. + +Like :mod:`flowx.discovery_insights`, there is **no LLM here**: the tool computes the recommendation +deterministically and only *validates and records* the agent-authored decision. The agent authors +**only** the decision (and an optional rationale); the library recomputes ``members``, +``recommended``, and both options' evidence on record so they can never drift from the inventory or +be faked. + +The recorded plan is bound to the inventory via :func:`flowx.discovery_insights.inventory_fingerprint` +-- a SHA-256 over the deterministic inventory base (the ``insights`` block excluded). That base is +exactly the structural signal that decides component membership and engine capability (pipelines, +control lineage, per-activity strategy, motifs); insights are advisory evidence for the agentic +option, not part of the binding. The write is atomic (temp file + ``os.replace``) and idempotent, so +re-recording the same decision against the same inventory rewrites byte-identical bytes. + +This is additive, opt-in routing metadata only: with no recorded plan, ``convert`` and ``package`` +behave exactly as today. Component computation reads ``lineage.control_edges``, which both ADF +(``ExecutePipeline``) and Airflow (``RunJob``) emit, so the artifact is source-neutral. +""" + +from __future__ import annotations + +import json +import os +from collections import Counter +from collections.abc import Iterable +from pathlib import Path +from typing import Any + +from flowx.discovery_insights import inventory_fingerprint +from flowx.models.conversion_plan import DECISIONS, SCHEMA_VERSION + +# The strategy value that marks an activity as individually engine-capable (a 1:1 deterministic +# translation). Any other value (``"agentic"`` / ``"unsupported"`` / missing) is a gap unless the +# activity is claimed by a detected motif -- the multi-activity capability signal from #64. +_DETERMINISTIC_STRATEGY = "deterministic" + +# The recorded conversion-plan artifact lives beside inventory.json under metadata/. +PLAN_FILENAME = "conversion_plan.json" + +# Authored top-level keys (everything else the library owns and rejects on input). +_PLAN_TOP_KEYS = {"components"} +_LIBRARY_TOP_KEYS = {"schema_version", "inventory_sha256", "findings"} +# Authored per-component keys vs the fields the library recomputes and rejects on input. +_COMPONENT_AUTHORED_KEYS = {"component_id", "members", "decision", "rationale"} +_COMPONENT_LIBRARY_KEYS = {"recommended", "options"} + + +def _pipeline_names(inventory: dict[str, Any]) -> set[str]: + """The set of real pipeline names in the inventory (the foreign-key domain). + + A local copy of the same projection :mod:`flowx.discovery_insights` uses, kept here so routing + does not depend on that module's private helpers. + """ + return { + str(pipeline["name"]) + for pipeline in inventory.get("pipelines", []) + if isinstance(pipeline, dict) and pipeline.get("name") is not None + } + + +def _control_edges(inventory: dict[str, Any]) -> Iterable[tuple[str, str, str, bool]]: + """Yield ``(source_workflow, target_workflow, via_task_key, resolved)`` for every control edge. + + Lineage is placed per pipeline (one block beside each pipeline's ``activities``), so every + pipeline's ``lineage.control_edges`` are gathered into one stream, preserving inventory order. + """ + for pipeline in inventory.get("pipelines", []): + if not isinstance(pipeline, dict): + continue + lineage = pipeline.get("lineage") or {} + for edge in lineage.get("control_edges", []): + if not isinstance(edge, dict): + continue + yield ( + str(edge.get("source_workflow") or ""), + str(edge.get("target_workflow") or ""), + str(edge.get("via_task_key") or ""), + bool(edge.get("resolved", True)), + ) + + +def build_components(inventory: dict[str, Any]) -> tuple[list[list[str]], list[str]]: + """Group pipelines into weak/undirected connected components over control lineage. + + Two pipelines share a component when a **resolved** control edge joins them in either direction + and both endpoints are real inventory pipelines. Isolated pipelines each form their own + singleton component. An edge whose callee is unresolved (``resolved`` is False) or names a + pipeline absent from the inventory is **not** used to join anything and is **not** silently + severed -- it is recorded as a finding so a partial export never quietly collapses two + components into one or drops a coupling. + + Args: + inventory: The discover ``inventory.json`` document (deterministic or enriched). + + Returns: + ``(components, findings)`` where ``components`` is a list of member lists -- each sorted, + the whole list ordered by first member -- so the output is deterministic and idempotent for + a given inventory; and ``findings`` is a sorted list of human-readable notes about + unresolved/dangling edges. + """ + names = _pipeline_names(inventory) + parent: dict[str, str] = {name: name for name in names} + + def find(node: str) -> str: + root = node + while parent[root] != root: + root = parent[root] + while parent[node] != root: + parent[node], node = root, parent[node] + return root + + def union(left: str, right: str) -> None: + left_root, right_root = find(left), find(right) + if left_root != right_root: + # Attach the lexicographically larger root under the smaller for a stable shape. + low, high = sorted((left_root, right_root)) + parent[high] = low + + findings: set[str] = set() + for source, target, via, resolved in _control_edges(inventory): + if not resolved or not target or target not in names: + shown_target = target or "" + findings.add( + f"unresolved control edge {source!r} -> {shown_target!r} (via {via!r}) " + f"kept as a finding, not severed; {source!r} is routed within its own component" + ) + continue + if source in names: + union(source, target) + + groups: dict[str, list[str]] = {} + for name in names: + groups.setdefault(find(name), []).append(name) + components = sorted((sorted(members) for members in groups.values()), key=lambda members: members) + return components, sorted(findings) + + +def _activities_by_pipeline(inventory: dict[str, Any]) -> dict[str, list[dict[str, Any]]]: + """Map each pipeline name to its inventory ``activities`` list (flattened, as emitted).""" + result: dict[str, list[dict[str, Any]]] = {} + for pipeline in inventory.get("pipelines", []): + if isinstance(pipeline, dict) and pipeline.get("name") is not None: + activities = pipeline.get("activities") or [] + result[str(pipeline["name"])] = [entry for entry in activities if isinstance(entry, dict)] + return result + + +def _motifs_by_pipeline(inventory: dict[str, Any]) -> dict[str, list[dict[str, Any]]]: + """Map each pipeline name to its additive ``motifs`` list (the #64 capability signal).""" + result: dict[str, list[dict[str, Any]]] = {} + for pipeline in inventory.get("pipelines", []): + if isinstance(pipeline, dict) and pipeline.get("name") is not None: + motifs = pipeline.get("motifs") or [] + result[str(pipeline["name"])] = [entry for entry in motifs if isinstance(entry, dict)] + return result + + +def _pipeline_insights(inventory: dict[str, Any]) -> dict[str, dict[str, Any]]: + """Map each pipeline name to its ``insights.pipeline_insights`` entry, when insights are present. + + Returns an empty map when the inventory has not been enriched, so the agentic option degrades + to an empty pattern list rather than failing. + """ + insights = inventory.get("insights") + if not isinstance(insights, dict): + return {} + result: dict[str, dict[str, Any]] = {} + for entry in insights.get("pipeline_insights", []): + if isinstance(entry, dict) and entry.get("pipeline") is not None: + result[str(entry["pipeline"])] = entry + return result + + +def _deterministic_option(members: list[str], inventory: dict[str, Any]) -> dict[str, Any]: + """Assess the deterministic (1:1 engine) conversion of a component. + + An activity is engine-capable when its ``strategy`` is ``"deterministic"`` or it is claimed by a + detected motif (present in some motif's ``member_task_keys``). ``capable`` is True only when the + whole component leaves no gap; ``uncovered`` cites each activity that keeps it from being fully + deterministic. + """ + activities_by_pipeline = _activities_by_pipeline(inventory) + motifs_by_pipeline = _motifs_by_pipeline(inventory) + + motif_ids: list[str] = [] + # Motif coverage is keyed by (pipeline, activity name), never bare name: a motif claiming a task + # in one pipeline must not mark an unrelated same-named task in another pipeline of the component + # as covered, which would silently hide a real gap. + covered_activities: set[tuple[str, str]] = set() + for pipeline in members: + for motif in motifs_by_pipeline.get(pipeline, []): + if motif.get("motif_id") is not None: + motif_ids.append(str(motif["motif_id"])) + covered_activities.update((pipeline, str(key)) for key in motif.get("member_task_keys") or []) + + counts = {"deterministic": 0, "agentic": 0, "unsupported": 0} + uncovered: list[dict[str, Any]] = [] + for pipeline in members: + for activity in activities_by_pipeline.get(pipeline, []): + strategy = activity.get("strategy") + bucket = strategy if strategy in ("deterministic", "agentic") else "unsupported" + counts[bucket] += 1 + name = activity.get("name") + if strategy != _DETERMINISTIC_STRATEGY and (pipeline, name) not in covered_activities: + uncovered.append( + { + "pipeline": pipeline, + "activity": name, + "type": activity.get("type"), + "strategy": strategy, + } + ) + return { + "capable": not uncovered, + "activity_counts": counts, + "motifs": sorted(set(motif_ids)), + "uncovered": uncovered, + } + + +# Neutral disclosure labels for the agentic option, keyed by release state. ``"ga"`` and ``"unknown"`` +# carry NO entry -- they are silent (we do not surface them, and ``"unknown"`` is treated exactly like +# ``"ga"``). ``"public_preview"`` is labelled production-ready per Databricks; ``"private_preview"`` / +# ``"beta"`` are stated as plain factual labels. This is disclosure, not a warning. +_RELEASE_STATE_LABELS: dict[str, str] = { + "public_preview": "Public Preview (production-ready)", + "private_preview": "Private Preview", + "beta": "Beta", +} + + +def _release_disclosure(pattern: dict[str, Any]) -> dict[str, Any] | None: + """Build a neutral release-state disclosure for one recommended pattern, or ``None``. + + Returns ``None`` for ``"ga"`` / ``"unknown"`` (both silent -- ``"unknown"`` is treated exactly like + ``"ga"``) and for a pattern with no ``release_state``. ``"public_preview"`` / ``"private_preview"`` + / ``"beta"`` each yield a factual state ``label`` -- ``public_preview`` noted as production-ready -- + with no warning framing. The cited source is appended when the insight carried one. + """ + release_state = pattern.get("release_state") + label = _RELEASE_STATE_LABELS.get(release_state) if isinstance(release_state, str) else None + if label is None: + return None + message = f"'{pattern.get('pattern')}' (pipeline '{pattern.get('pipeline')}'): {label}." + source = pattern.get("release_state_source") + if isinstance(source, str) and source.strip(): + message = f"{message} Source: {source.strip()}" + return { + "pipeline": pattern.get("pipeline"), + "pattern": pattern.get("pattern"), + "release_state": release_state, + "label": label, + "message": message, + } + + +def _agentic_option(members: list[str], inventory: dict[str, Any]) -> dict[str, Any]: + """Surface the agent-authored recommended patterns for a component as a first-class option. + + Draws every ``recommended_patterns`` entry from the member pipelines' insights, tagging each with + its pipeline (the per-pattern ``release_state`` / ``release_state_source`` ride along in the + ``{**pattern}`` spread), and computes two signals for the user: + + * ``has_simplification`` -- any pattern is a ``simplification_pattern`` (a distinctive + re-architecture such as a multi-pipeline -> Lakeflow Connect collapse), surfaced prominently. + * ``release_disclosures`` -- a neutral, per-pattern disclosure of any non-silent ``release_state`` + (:data:`~flowx.models.conversion_plan.DISCLOSED_RELEASE_STATES`): ``public_preview`` labelled + production-ready, ``private_preview`` / ``beta`` stated as plain labels. ``ga`` and ``unknown`` + add nothing (silent). Factual labelling, not a warning. + """ + insights_by_pipeline = _pipeline_insights(inventory) + recommended_patterns: list[dict[str, Any]] = [] + for pipeline in members: + insight = insights_by_pipeline.get(pipeline) + if not insight: + continue + for pattern in insight.get("recommended_patterns") or []: + if isinstance(pattern, dict): + recommended_patterns.append({"pipeline": pipeline, **pattern}) + has_simplification = any(pattern.get("simplification_pattern") for pattern in recommended_patterns) + release_disclosures: list[dict[str, Any]] = [] + for pattern in recommended_patterns: + disclosure = _release_disclosure(pattern) + if disclosure is not None: + release_disclosures.append(disclosure) + return { + "recommended_patterns": recommended_patterns, + "has_simplification": has_simplification, + "release_disclosures": release_disclosures, + } + + +def recommend_component(members: list[str], inventory: dict[str, Any]) -> tuple[str, dict[str, Any]]: + """Compute the recommended route and both conversion options for one component. + + Args: + members: The component's pipeline names. + inventory: The discover ``inventory.json`` document (deterministic or enriched). + + Returns: + ``(recommended, options)`` where ``recommended`` is ``"deterministic"`` when the whole + component is engine-capable, else ``"agentic"``; and ``options`` carries the ``deterministic`` + and ``agentic`` peers as first-class entries. + """ + deterministic = _deterministic_option(members, inventory) + agentic = _agentic_option(members, inventory) + recommended = _DETERMINISTIC_STRATEGY if deterministic["capable"] else "agentic" + return recommended, {"deterministic": deterministic, "agentic": agentic} + + +def build_recommendation(inventory: dict[str, Any]) -> dict[str, Any]: + """Compute the full routing recommendation over every component in the inventory. + + Returns a dict with ``components`` (each carrying ``component_id``, sorted ``members``, + ``recommended`` and both ``options``), the ``findings`` from component computation, and a + ``default_plan`` that proposes ``decision == recommended`` for every component -- ready to hand + straight to :func:`record_plan` when the user accepts the recommendations wholesale, or to edit + per component for overrides. + """ + components, findings = build_components(inventory) + component_entries: list[dict[str, Any]] = [] + default_plan_components: list[dict[str, Any]] = [] + for index, members in enumerate(components, start=1): + component_id = f"component-{index}" + recommended, options = recommend_component(members, inventory) + component_entries.append( + {"component_id": component_id, "members": members, "recommended": recommended, "options": options} + ) + default_plan_components.append({"component_id": component_id, "members": members, "decision": recommended}) + return { + "components": component_entries, + "findings": findings, + "default_plan": {"components": default_plan_components}, + } + + +# --------------------------------------------------------------------------- # +# Validation. All violations are collected (never fail-fast) so the authoring +# agent can fix every problem in one pass. +# --------------------------------------------------------------------------- # + + +def _components_by_id(inventory: dict[str, Any]) -> dict[str, list[str]]: + """Computed components keyed by their library id (``component-``).""" + components, _ = build_components(inventory) + return {f"component-{index}": members for index, members in enumerate(components, start=1)} + + +def validate_plan(raw: Any, inventory: dict[str, Any]) -> list[str]: + """Validate an authored conversion plan against the inventory's connected components. + + Returns a list of human-readable violation strings; an empty list means the plan is valid. Never + raises on a malformed payload. The rules: + + * only the authored top-level key ``components`` is allowed (library-owned keys are rejected with + a hint), and each component entry may carry only the authored fields (``recommended`` / + ``options`` are library-computed and rejected on input); + * ``decision`` must be one of the known routes and ``rationale`` (when present) a non-empty string; + * every member must be a real inventory pipeline, and a component's ``members`` must exactly match + one computed connected component -- so a decision can never split a component or span two; + * the plan is a **bijection** over components: every component is decided exactly once (no + partial plan, no duplicate/conflicting decisions). + """ + if not isinstance(raw, dict): + return [f"conversion plan must be a JSON object, got {type(raw).__name__}"] + + violations: list[str] = [] + for key in sorted(set(raw) - _PLAN_TOP_KEYS): + hint = " (set by the library, not the author)" if key in _LIBRARY_TOP_KEYS else "" + violations.append(f"unknown top-level key: {key!r}{hint}") + + components = raw.get("components") + if not isinstance(components, list): + violations.append("'components' must be a list") + return violations + + names = _pipeline_names(inventory) + computed_by_id = _components_by_id(inventory) + computed_by_members = {frozenset(members): component_id for component_id, members in computed_by_id.items()} + + decided_ids: list[str] = [] + for index, component in enumerate(components): + loc = f"components[{index}]" + matched_id = _validate_component_entry(component, loc, names, computed_by_members, violations) + if matched_id is not None: + decided_ids.append(matched_id) + + for component_id, count in Counter(decided_ids).items(): + if count > 1: + violations.append( + f"component {component_id!r} is decided {count} times; each component needs exactly one decision" + ) + decided = set(decided_ids) + for component_id, members in computed_by_id.items(): + if component_id not in decided: + violations.append(f"component {component_id!r} ({members}) has no decision; every component must be routed") + return violations + + +def _validate_component_entry( + component: Any, + loc: str, + names: set[str], + computed_by_members: dict[frozenset[str], str], + violations: list[str], +) -> str | None: + """Validate one authored component entry, appending problems; return the matched component id. + + Returns the computed ``component-`` id this entry decides when its ``members`` exactly match a + connected component (so the caller can enforce the bijection), else ``None``. + """ + if not isinstance(component, dict): + violations.append(f"{loc} must be an object") + return None + + for key in sorted(set(component) - _COMPONENT_AUTHORED_KEYS): + hint = " (set by the library, not the author)" if key in _COMPONENT_LIBRARY_KEYS else "" + violations.append(f"{loc}: unknown field {key!r}{hint}") + + decision = component.get("decision") + if decision not in DECISIONS: + allowed = ", ".join(repr(value) for value in DECISIONS) + violations.append(f"{loc}: 'decision' must be one of {{{allowed}}}, got {decision!r}") + + rationale = component.get("rationale") + if rationale is not None and (not isinstance(rationale, str) or not rationale.strip()): + violations.append(f"{loc}: 'rationale' must be a non-empty string when present") + + component_id = component.get("component_id") + if not isinstance(component_id, str) or not component_id: + violations.append(f"{loc}: 'component_id' must be a non-empty string") + + members = component.get("members") + if not isinstance(members, list) or not all(isinstance(member, str) for member in members): + violations.append(f"{loc}: 'members' must be a list of pipeline names") + return None + + for member in members: + if member not in names: + violations.append(f"{loc}: pipeline {member!r} not in inventory") + + # Reject duplicate members explicitly: a frozenset match would collapse ["a", "a"] to {"a"} and + # wrongly accept it as the component {"a"}, breaking the members-match / bijection contract. + duplicates = sorted({member for member in members if members.count(member) > 1}) + if duplicates: + violations.append(f"{loc}: duplicate members {duplicates}; list each pipeline once") + return None + + matched_id = computed_by_members.get(frozenset(members)) + if matched_id is None: + violations.append( + f"{loc}: members {sorted(members)} do not form a connected component " + f"(they split or span computed components); route each component as a whole" + ) + return None + if isinstance(component_id, str) and component_id and component_id != matched_id: + violations.append( + f"{loc}: component_id {component_id!r} does not match the component for these members " + f"(expected {matched_id!r})" + ) + return matched_id + + +# --------------------------------------------------------------------------- # +# Loading, recording, and the atomic idempotent write. +# --------------------------------------------------------------------------- # + + +def load_plan(*, plan: dict[str, Any] | None = None, plan_path: Path | None = None) -> dict[str, Any]: + """Return the raw authored plan dict from exactly one source (inline or file). + + Raises: + ValueError: if neither or both sources are provided. + """ + if (plan is None) == (plan_path is None): + raise ValueError("provide exactly one of 'plan' (inline dict) or 'plan_path'") + if plan is not None: + return plan + assert plan_path is not None # guaranteed by the guard above + return json.loads(Path(plan_path).read_text(encoding="utf-8")) + + +def build_plan_document(inventory: dict[str, Any], raw: dict[str, Any]) -> dict[str, Any]: + """Build the recorded plan document from a validated authored plan. + + The library recomputes ``members`` / ``recommended`` / both ``options`` and overlays only the + authored ``decision`` (and optional ``rationale``) per component, so the recorded facts cannot + drift from the inventory. Stamps the library-owned ``schema_version`` and ``inventory_sha256``. + Does not mutate the inputs and performs no I/O. + """ + recommendation = build_recommendation(inventory) + computed_by_id = _components_by_id(inventory) + members_to_id = {frozenset(members): component_id for component_id, members in computed_by_id.items()} + + authored_by_id: dict[str, dict[str, Any]] = {} + for component in raw.get("components", []): + component_id = members_to_id[frozenset(component["members"])] + authored_by_id[component_id] = component + + components_out: list[dict[str, Any]] = [] + for entry in recommendation["components"]: + component_id = entry["component_id"] + authored = authored_by_id[component_id] + recorded: dict[str, Any] = { + "component_id": component_id, + "members": entry["members"], + "recommended": entry["recommended"], + "decision": authored["decision"], + "options": entry["options"], + } + rationale = authored.get("rationale") + if rationale is not None: + recorded["rationale"] = rationale + components_out.append(recorded) + + return { + "schema_version": SCHEMA_VERSION, + "inventory_sha256": inventory_fingerprint(inventory), + "components": components_out, + "findings": recommendation["findings"], + } + + +def _write_plan_atomic(path: Path, document: dict[str, Any]) -> None: + """Write the plan JSON atomically (temp file + ``os.replace``), matching the inventory formatting.""" + path.parent.mkdir(parents=True, exist_ok=True) + temporary = path.with_name(f".{path.name}.tmp") + temporary.write_text(json.dumps(document, indent=2), encoding="utf-8") + os.replace(temporary, path) + + +def record_plan( + output_dir: Path, + *, + plan: dict[str, Any] | None = None, + plan_path: Path | None = None, +) -> dict[str, Any]: + """Validate an authored plan against the inventory, then record it on success. + + Reads ``/metadata/inventory.json``, validates the authored plan, and -- only when + there are no violations -- writes ``/metadata/conversion_plan.json`` atomically. The + inventory file is never touched. Provide the authored plan via exactly one of ``plan`` (inline + dict) or ``plan_path`` (a JSON file). + + Returns ``{"ok", "violations", "inventory_sha256", "components", "findings"}``. ``ok`` is + ``False`` (and no plan written) when there are violations. + + Raises: + FileNotFoundError: when ``inventory.json`` does not exist (run discover first). + ValueError: when neither or both plan sources are provided, or the inventory is not a JSON + object. + """ + inventory_path = Path(output_dir) / "metadata" / "inventory.json" + if not inventory_path.exists(): + raise FileNotFoundError(f"No inventory.json under {inventory_path.parent}; run the discover phase first.") + inventory = json.loads(inventory_path.read_text(encoding="utf-8")) + if not isinstance(inventory, dict): + raise ValueError(f"inventory.json must contain a JSON object, got {type(inventory).__name__}") + + raw = load_plan(plan=plan, plan_path=plan_path) + violations = validate_plan(raw, inventory) + if violations: + return {"ok": False, "violations": violations, "components": 0} + + document = build_plan_document(inventory, raw) + _write_plan_atomic(inventory_path.with_name(PLAN_FILENAME), document) + return { + "ok": True, + "violations": [], + "inventory_sha256": document["inventory_sha256"], + "components": len(document["components"]), + "findings": len(document["findings"]), + } diff --git a/tests/unit/test_cli_route_agentic.py b/tests/unit/test_cli_route_agentic.py new file mode 100644 index 0000000..a43081e --- /dev/null +++ b/tests/unit/test_cli_route_agentic.py @@ -0,0 +1,332 @@ +"""Tests for the reshaped ``route`` CLI and the ``fill-agentic`` combine surface. + +``route`` is now one command: with no decision on a non-TTY it emits the recommendation (a dry run +agents read first); with a decision (``--plan-path``, ``--plan-path -`` for stdin, or an interactive +TTY prompt) it records the fingerprint-bound plan AND edits the report for the routed-agentic groups. +``fill-agentic combine`` performs the cross-pipeline pipeline-grain fill, validating structurally +before writing. +""" + +from __future__ import annotations + +import io +import json +from pathlib import Path +from typing import Any + +import pytest + +from flowx.adapter.__main__ import main as adapter_cli_main +from flowx.discovery_inventory import STRATEGY_PROPERTY, build_source_inventory +from flowx.models.discovery import CONCEPT_NOTEBOOK, SourceGraph, SourceNode +from flowx.models.ir import ControlEdge, Lineage +from flowx.route_agentic import GAPS_FILENAME, REPORT_FILENAME, WORK_DIRNAME + + +def _node(task_key: str, native_type: str, *, strategy: str = "deterministic") -> SourceNode: + return SourceNode( + source_id=task_key, + task_key=task_key, + concept=CONCEPT_NOTEBOOK, + source="unit", + name=task_key, + native_type=native_type, + properties={STRATEGY_PROPERTY: strategy}, + raw={"name": task_key, "type": native_type}, + ) + + +def _inventory() -> dict[str, Any]: + """One component 'parent -> child' plus a standalone 'solo'.""" + parent = SourceGraph( + name="parent", + source="unit", + tasks=[_node("call_child", "ExecutePipeline")], + lineage=Lineage( + control_edges=[ControlEdge(source_workflow="parent", target_workflow="child", via_task_key="call_child")] + ), + ) + child = SourceGraph(name="child", source="unit", tasks=[_node("copy_orders", "Copy")]) + solo = SourceGraph(name="solo", source="unit", tasks=[_node("load", "Notebook")]) + return build_source_inventory([parent, child, solo], source="unit", source_dir="/tmp/src") + + +def _report() -> dict[str, Any]: + return { + "pipelines": [ + {"name": "parent", "tasks": [{"name": "call_child", "task_key": "call_child", "type": "CopyActivity"}]}, + {"name": "child", "tasks": [{"name": "copy_orders", "task_key": "copy_orders", "type": "CopyActivity"}]}, + { + "name": "solo", + "tasks": [ + {"name": "load", "task_key": "load", "type": "NotebookActivity", "notebook_path": "/x"}, + ], + }, + ] + } + + +def _setup(output_dir: Path) -> None: + metadata = output_dir / "metadata" + metadata.mkdir(parents=True, exist_ok=True) + (metadata / "inventory.json").write_text(json.dumps(_inventory(), indent=2), encoding="utf-8") + work = output_dir / WORK_DIRNAME + work.mkdir(parents=True, exist_ok=True) + (work / REPORT_FILENAME).write_text(json.dumps(_report(), indent=2), encoding="utf-8") + (work / GAPS_FILENAME).write_text("[]", encoding="utf-8") + + +def _agentic_plan() -> dict[str, Any]: + """Route the parent<->child component agentic; leave solo deterministic.""" + return { + "components": [ + {"component_id": "component-1", "members": ["child", "parent"], "decision": "agentic"}, + {"component_id": "component-2", "members": ["solo"], "decision": "deterministic"}, + ] + } + + +def test_route_without_a_decision_on_a_non_tty_emits_the_recommendation( + tmp_path: Path, capsys: pytest.CaptureFixture[str], monkeypatch: pytest.MonkeyPatch +) -> None: + _setup(tmp_path) + monkeypatch.setattr("sys.stdin", io.StringIO("")) # not a TTY + code = adapter_cli_main(["route", "--output-dir", str(tmp_path)]) + assert code == 0 + payload = json.loads(capsys.readouterr().out) + assert [c["component_id"] for c in payload["components"]] == ["component-1", "component-2"] + assert "default_plan" in payload + # No decision => report untouched. + report = json.loads((tmp_path / WORK_DIRNAME / REPORT_FILENAME).read_text(encoding="utf-8")) + assert all(pipeline["tasks"][0]["type"] != "PlaceholderActivity" for pipeline in report["pipelines"]) + + +def test_route_with_plan_path_records_and_edits_the_report(tmp_path: Path, capsys: pytest.CaptureFixture[str]) -> None: + _setup(tmp_path) + plan_path = tmp_path / "plan.json" + plan_path.write_text(json.dumps(_agentic_plan()), encoding="utf-8") + code = adapter_cli_main(["route", "--output-dir", str(tmp_path), "--plan-path", str(plan_path)]) + assert code == 0 + payload = json.loads(capsys.readouterr().out) + assert payload["ok"] is True + assert payload["edit"]["agentic_pipelines"] == ["child", "parent"] + + # The recorded plan is written (fingerprint binding preserved). + assert (tmp_path / "metadata" / "conversion_plan.json").exists() + # The routed-agentic pipelines were placeholdered; the deterministic 'solo' is untouched. + report = json.loads((tmp_path / WORK_DIRNAME / REPORT_FILENAME).read_text(encoding="utf-8")) + by_name = {pipeline["name"]: pipeline for pipeline in report["pipelines"]} + assert by_name["parent"]["tasks"][0]["type"] == "PlaceholderActivity" + assert by_name["child"]["tasks"][0]["type"] == "PlaceholderActivity" + assert by_name["solo"]["tasks"][0]["type"] == "NotebookActivity" + gaps = json.loads((tmp_path / WORK_DIRNAME / GAPS_FILENAME).read_text(encoding="utf-8")) + assert sorted(gap["pipeline"] for gap in gaps) == ["child", "parent"] + + +def test_route_reads_a_plan_from_stdin_when_plan_path_is_dash( + tmp_path: Path, capsys: pytest.CaptureFixture[str], monkeypatch: pytest.MonkeyPatch +) -> None: + _setup(tmp_path) + monkeypatch.setattr("sys.stdin", io.StringIO(json.dumps(_agentic_plan()))) + code = adapter_cli_main(["route", "--output-dir", str(tmp_path), "--plan-path", "-"]) + assert code == 0 + payload = json.loads(capsys.readouterr().out) + assert payload["ok"] is True and payload["edit"]["agentic_pipelines"] == ["child", "parent"] + + +def test_route_interactive_prompt_records_the_users_decision( + tmp_path: Path, capsys: pytest.CaptureFixture[str], monkeypatch: pytest.MonkeyPatch +) -> None: + _setup(tmp_path) + + class _Tty(io.StringIO): + def isatty(self) -> bool: + return True + + monkeypatch.setattr("sys.stdin", _Tty("")) + answers = iter(["a", ""]) # component-1 -> agentic; component-2 -> accept recommendation (deterministic) + monkeypatch.setattr("builtins.input", lambda _prompt="": next(answers)) + + code = adapter_cli_main(["route", "--output-dir", str(tmp_path)]) + assert code == 0 + payload = json.loads(capsys.readouterr().out) + assert payload["ok"] is True and payload["edit"]["agentic_pipelines"] == ["child", "parent"] + + +def test_route_validation_failure_returns_1_and_leaves_report_untouched( + tmp_path: Path, capsys: pytest.CaptureFixture[str] +) -> None: + _setup(tmp_path) + report_before = (tmp_path / WORK_DIRNAME / REPORT_FILENAME).read_bytes() + bad = _agentic_plan() + bad["components"].pop() # partial plan: component-2 undecided + plan_path = tmp_path / "plan.json" + plan_path.write_text(json.dumps(bad), encoding="utf-8") + code = adapter_cli_main(["route", "--output-dir", str(tmp_path), "--plan-path", str(plan_path)]) + assert code == 1 + payload = json.loads(capsys.readouterr().out) + assert payload["ok"] is False and payload["violations"] + assert (tmp_path / WORK_DIRNAME / REPORT_FILENAME).read_bytes() == report_before + + +def test_route_with_a_missing_plan_file_fails_cleanly(tmp_path: Path) -> None: + _setup(tmp_path) + code = adapter_cli_main(["route", "--output-dir", str(tmp_path), "--plan-path", str(tmp_path / "nope.json")]) + assert code == 1 + + +def test_route_errors_when_report_missing_and_no_source_to_trigger_convert( + tmp_path: Path, capsys: pytest.CaptureFixture[str] +) -> None: + metadata = tmp_path / "metadata" + metadata.mkdir(parents=True) + (metadata / "inventory.json").write_text(json.dumps(_inventory(), indent=2), encoding="utf-8") + plan_path = tmp_path / "plan.json" + plan_path.write_text(json.dumps(_agentic_plan()), encoding="utf-8") + code = adapter_cli_main(["route", "--output-dir", str(tmp_path), "--plan-path", str(plan_path)]) + assert code == 1 + + +# --------------------------------------------------------------------------- # +# fill-agentic combine. +# --------------------------------------------------------------------------- # + + +def _lfc_pipeline() -> dict[str, Any]: + definition = {"name": "orders_ingestion", "catalog": "${var.catalog}", "target": "${var.schema}"} + return { + "name": "orders_lfc", + "tags": {"source": "adf"}, + "tasks": [ + { + "name": "Ingest orders", + "task_key": "ingest_orders", + "type": "AgenticComponentActivity", + "files": [], + "resources": [{"resource_key": "orders_ingestion", "definition": definition}], + "task": {"pipeline_task": {"pipeline_id": "${resources.pipelines.orders_ingestion.id}"}}, + } + ], + } + + +def _record_agentic_plan(tmp_path: Path) -> None: + """Route the parent<->child component agentic and record the plan so combine can bind to it.""" + plan_path = tmp_path / "route_plan.json" + plan_path.write_text(json.dumps(_agentic_plan()), encoding="utf-8") + assert adapter_cli_main(["route", "--output-dir", str(tmp_path), "--plan-path", str(plan_path)]) == 0 + + +def _run_combine(tmp_path: Path, members: str, authored: list[dict[str, Any]]) -> int: + pipelines_path = tmp_path / "authored.json" + pipelines_path.write_text(json.dumps(authored), encoding="utf-8") + return adapter_cli_main( + [ + "fill-agentic", + "combine", + "--output-dir", + str(tmp_path), + "--members", + members, + "--pipelines-path", + str(pipelines_path), + ] + ) + + +def test_fill_agentic_combine_writes_merged_report(tmp_path: Path, capsys: pytest.CaptureFixture[str]) -> None: + _setup(tmp_path) + _record_agentic_plan(tmp_path) + capsys.readouterr() # drop the route-record output so only the combine JSON remains + code = _run_combine(tmp_path, "child,parent", [_lfc_pipeline()]) + assert code == 0 + payload = json.loads(capsys.readouterr().out) + assert payload["ok"] is True + report = json.loads((tmp_path / WORK_DIRNAME / REPORT_FILENAME).read_text(encoding="utf-8")) + names = sorted(pipeline["name"] for pipeline in report["pipelines"]) + assert names == ["orders_lfc", "solo"] + + +def test_fill_agentic_combine_rejects_dangling_reference(tmp_path: Path, capsys: pytest.CaptureFixture[str]) -> None: + _setup(tmp_path) + _record_agentic_plan(tmp_path) + capsys.readouterr() # drop the route-record output so only the combine JSON remains + report_before = (tmp_path / WORK_DIRNAME / REPORT_FILENAME).read_bytes() + dangling = _lfc_pipeline() + dangling["tasks"][0]["task"] = {"pipeline_task": {"pipeline_id": "${resources.pipelines.ghost.id}"}} + code = _run_combine(tmp_path, "child,parent", [dangling]) + assert code == 1 + payload = json.loads(capsys.readouterr().out) + assert payload["ok"] is False and payload["violations"] + assert (tmp_path / WORK_DIRNAME / REPORT_FILENAME).read_bytes() == report_before + + +def test_fill_agentic_combine_rejects_a_deterministic_member_set( + tmp_path: Path, capsys: pytest.CaptureFixture[str] +) -> None: + _setup(tmp_path) + _record_agentic_plan(tmp_path) # component-1 (child, parent) agentic; solo is deterministic + capsys.readouterr() # drop the route-record output so only the combine JSON remains + code = _run_combine(tmp_path, "solo", [_lfc_pipeline()]) + assert code == 1 + payload = json.loads(capsys.readouterr().out) + assert payload["ok"] is False and payload["error"] + + +def test_fill_agentic_has_no_no_validate_flag(tmp_path: Path) -> None: + # The validation bypass must not exist as a CLI surface; argparse rejects the unknown flag. + with pytest.raises(SystemExit) as excinfo: + adapter_cli_main( + [ + "fill-agentic", + "combine", + "--output-dir", + str(tmp_path), + "--members", + "child,parent", + "--pipelines-path", + str(tmp_path / "x.json"), + "--no-validate", + ] + ) + assert excinfo.value.code == 2 + + +def test_route_triggers_convert_when_report_absent( + tmp_path: Path, capsys: pytest.CaptureFixture[str], monkeypatch: pytest.MonkeyPatch +) -> None: + # Inventory present, report absent: with --source/--source-path, route triggers convert. Stub the + # phase runner to write the report the trigger would have produced, then assert route records+edits. + metadata = tmp_path / "metadata" + metadata.mkdir(parents=True) + (metadata / "inventory.json").write_text(json.dumps(_inventory(), indent=2), encoding="utf-8") + + triggered: list[list[str]] = [] + + def fake_run_phase(phase: str, forward: list[str]) -> int: + triggered.append([phase, *forward]) + work = tmp_path / WORK_DIRNAME + work.mkdir(parents=True, exist_ok=True) + (work / REPORT_FILENAME).write_text(json.dumps(_report(), indent=2), encoding="utf-8") + return 0 + + monkeypatch.setattr("flowx.adapter.__main__._run_phase", fake_run_phase) + plan_path = tmp_path / "plan.json" + plan_path.write_text(json.dumps(_agentic_plan()), encoding="utf-8") + code = adapter_cli_main( + [ + "route", + "--output-dir", + str(tmp_path), + "--plan-path", + str(plan_path), + "--source", + "adf", + "--source-path", + str(tmp_path / "adf_src"), + ] + ) + assert code == 0 + assert triggered and triggered[0][0] == "convert" + payload = json.loads(capsys.readouterr().out) + assert payload["ok"] is True and payload["edit"]["agentic_pipelines"] == ["child", "parent"] diff --git a/tests/unit/test_mcp_route.py b/tests/unit/test_mcp_route.py new file mode 100644 index 0000000..1dfd663 --- /dev/null +++ b/tests/unit/test_mcp_route.py @@ -0,0 +1,157 @@ +"""Tests that the reshaped MCP ``route`` / ``fill_agentic`` commands forward to the adapter CLI. + +``route`` is one command: no plan => the adapter emits the recommendation; a plan (inline or path) +=> the adapter records + edits via ``--plan-path``. ``fill_agentic`` performs the cross-pipeline +combine fill. +""" + +from __future__ import annotations + +import json +from pathlib import Path +from typing import Any + +import pytest + +pytest.importorskip("mcp") + +from flowx.mcp import runner, server # noqa: E402 + + +class _Result: + ok = True + returncode = 0 + + def __init__(self, stdout: str) -> None: + self.stdout = stdout + self.stderr = "" + + def as_dict(self) -> dict[str, Any]: + return {"returncode": 0, "stdout": self.stdout, "stderr": ""} + + +@pytest.fixture +def captured(monkeypatch): + calls: list[list[str]] = [] + + def fake_run_adapter(args, **_kwargs): + calls.append([str(a) for a in args]) + if args and args[0] == "route" and "--plan-path" not in [str(a) for a in args]: + return _Result(json.dumps({"components": [{"component_id": "component-1"}], "default_plan": {}})) + return _Result(json.dumps({"ok": True, "violations": [], "components": 1, "edit": {"agentic_pipelines": []}})) + + monkeypatch.setattr(runner, "run_adapter", fake_run_adapter) + return calls + + +def test_route_without_a_plan_emits_the_recommendation(captured, tmp_path: Path) -> None: + out = server._cmd_route({"output_dir": str(tmp_path)}) + assert out["ok"] is True + assert out["result"]["components"][0]["component_id"] == "component-1" + argv = captured[0] + assert argv[0] == "route" and "--plan-path" not in argv + assert "--output-dir" in argv + + +def test_route_inline_plan_is_staged_and_forwarded_as_plan_path(captured, tmp_path: Path) -> None: + plan = {"components": [{"component_id": "component-1", "members": ["a"], "decision": "agentic"}]} + out = server._cmd_route({"output_dir": str(tmp_path), "plan": plan}) + assert out["ok"] is True + argv = captured[0] + assert argv[0] == "route" + assert "--plan-path" in argv # inline plan staged to a real temp file + + +def test_route_forwards_an_explicit_plan_path(captured, tmp_path: Path) -> None: + plan_path = tmp_path / "plan.json" + plan_path.write_text("{}", encoding="utf-8") + server._cmd_route({"output_dir": str(tmp_path), "plan_path": str(plan_path)}) + argv = captured[0] + assert argv[argv.index("--plan-path") + 1] == str(plan_path) + + +def test_route_rejects_both_plan_and_plan_path(tmp_path: Path) -> None: + out = server._cmd_route({"output_dir": str(tmp_path), "plan": {}, "plan_path": "x.json"}) + assert out["ok"] is False and "at most one" in out["error"] + + +def test_route_forwards_source_so_convert_can_be_triggered(captured, tmp_path: Path) -> None: + # Parity with the CLI: MCP route forwards source + source-path so the CLI can trigger convert when + # the report is absent. Recommend path (no plan). + server._cmd_route({"output_dir": str(tmp_path), "source": "adf", "adf_source_path": "/tmp/adf"}) + argv = captured[0] + assert argv[argv.index("--source") + 1] == "adf" + assert argv[argv.index("--source-path") + 1] == "/tmp/adf" + + +def test_route_forwards_source_on_the_record_path(captured, tmp_path: Path) -> None: + plan = {"components": [{"component_id": "component-1", "members": ["a"], "decision": "agentic"}]} + server._cmd_route({"output_dir": str(tmp_path), "plan": plan, "source": "adf", "adf_source_path": "/tmp/adf"}) + argv = captured[0] + assert "--plan-path" in argv + assert argv[argv.index("--source") + 1] == "adf" + assert argv[argv.index("--source-path") + 1] == "/tmp/adf" + + +def test_route_without_source_forwards_no_source_flags(captured, tmp_path: Path) -> None: + server._cmd_route({"output_dir": str(tmp_path)}) + assert "--source" not in captured[0] + + +def test_route_registered_in_command_map() -> None: + assert "route" in server._COMMANDS + + +# --------------------------------------------------------------------------- # +# fill_agentic (combine). +# --------------------------------------------------------------------------- # + + +@pytest.fixture +def captured_fill(monkeypatch): + calls: list[list[str]] = [] + + def fake_run_adapter(args, **_kwargs): + calls.append([str(a) for a in args]) + return _Result(json.dumps({"ok": True, "violations": [], "pipelines": 1})) + + monkeypatch.setattr(runner, "run_adapter", fake_run_adapter) + return calls + + +def test_fill_agentic_inline_pipelines_are_staged(captured_fill, tmp_path: Path) -> None: + out = server._cmd_fill_agentic( + {"output_dir": str(tmp_path), "members": ["parent", "child"], "pipelines": [{"name": "lfc", "tasks": []}]} + ) + assert out["ok"] is True + argv = captured_fill[0] + assert argv[0] == "fill-agentic" and argv[1] == "combine" + assert argv[argv.index("--members") + 1] == "parent,child" + assert "--pipelines-path" in argv + + +def test_fill_agentic_forwards_members_string_and_path(captured_fill, tmp_path: Path) -> None: + pipelines_path = tmp_path / "authored.json" + pipelines_path.write_text("[]", encoding="utf-8") + server._cmd_fill_agentic({"output_dir": str(tmp_path), "members": "a,b", "pipelines_path": str(pipelines_path)}) + argv = captured_fill[0] + assert argv[argv.index("--members") + 1] == "a,b" + assert argv[argv.index("--pipelines-path") + 1] == str(pipelines_path) + + +def test_fill_agentic_requires_members(tmp_path: Path) -> None: + out = server._cmd_fill_agentic({"output_dir": str(tmp_path), "pipelines": []}) + assert out["ok"] is False and "members" in out["error"] + + +def test_fill_agentic_requires_exactly_one_pipelines_source(tmp_path: Path) -> None: + both = server._cmd_fill_agentic( + {"output_dir": str(tmp_path), "members": ["a"], "pipelines": [], "pipelines_path": "x.json"} + ) + neither = server._cmd_fill_agentic({"output_dir": str(tmp_path), "members": ["a"]}) + assert both["ok"] is False and "exactly one" in both["error"] + assert neither["ok"] is False and "exactly one" in neither["error"] + + +def test_fill_agentic_registered_in_command_map() -> None: + assert "fill_agentic" in server._COMMANDS diff --git a/tests/unit/test_route_agentic_flow.py b/tests/unit/test_route_agentic_flow.py new file mode 100644 index 0000000..65e5c0a --- /dev/null +++ b/tests/unit/test_route_agentic_flow.py @@ -0,0 +1,621 @@ +"""Tests for the Phase-1 in-engine agentic conversion flow (route -> alter -> fill). + +The confirmed flow, built here: + +* ``route`` groups pipelines by connected component (reusing :mod:`flowx.routing`), takes a + per-component deterministic/agentic decision, records the fingerprint-bound conversion plan, and + then **edits** the deterministic ``translation_report.json``: for every pipeline in an + agentic-routed component it removes the deterministic tasks, replaces them with + ``PlaceholderActivity`` entries, and appends one ``AgenticGap`` per task to ``gaps.json``. Pipelines + in deterministic components are left byte-identical, and with no agentic decision the report and + gaps are untouched (the non-breaking guarantee). +* the agent then authors the fill. **Per-pipeline** agentic reuses the existing name-matched + :func:`flowx.ir_serde.merge_agentic_results`. **Cross-pipeline COMBINE** (N pipelines -> M, e.g. one + Lakeflow Connect pipeline) uses a new pipeline-grain fill that swaps the routed group's pipelines + for the agent-authored pipeline(s), which carry ``AgenticComponentActivity`` nodes. +* the merged report is validated structurally via the existing + :func:`flowx.validate.bundle_invariants.check_bundle_dir` (unique keys, no dangling deps, acyclic, + no dangling pipeline/run_job references). +""" + +from __future__ import annotations + +import copy +import inspect +import json +from pathlib import Path +from typing import Any + +from flowx.discovery_inventory import STRATEGY_PROPERTY, build_source_inventory +from flowx.ir_serde import merge_agentic_results +from flowx.models.discovery import CONCEPT_NOTEBOOK, SourceGraph, SourceNode +from flowx.models.ir import ControlEdge, Lineage +from flowx.route_agentic import ( + GAPS_FILENAME, + REPORT_FILENAME, + WORK_DIRNAME, + agentic_pipeline_names, + alter_report, + apply_combine_fill, + apply_plan_to_report, + combine_group_fill, + prompt_for_decisions, + validate_report_structurally, +) +from flowx.routing import record_plan + +# --------------------------------------------------------------------------- # +# Fixtures. +# --------------------------------------------------------------------------- # + + +def _notebook_task(name: str, task_key: str, path: str = "/Workspace/Shared/x") -> dict[str, Any]: + return {"name": name, "task_key": task_key, "type": "NotebookActivity", "notebook_path": path} + + +def _copy_task(name: str, task_key: str) -> dict[str, Any]: + return {"name": name, "task_key": task_key, "type": "CopyActivity"} + + +def _report_two_pipelines() -> dict[str, Any]: + """A 'parent' pipeline (one Copy) and a 'child' pipeline (one Notebook).""" + return { + "pipelines": [ + {"name": "parent", "tasks": [_copy_task("Extract", "extract")]}, + {"name": "child", "tasks": [_notebook_task("Load", "load")]}, + ] + } + + +def _plan(*, parent: str = "agentic", child: str = "deterministic") -> dict[str, Any]: + """An authored/recorded plan: 'parent' and 'child' each their own component.""" + return { + "components": [ + {"component_id": "component-1", "members": ["parent"], "decision": parent}, + {"component_id": "component-2", "members": ["child"], "decision": child}, + ] + } + + +def _recommendation() -> dict[str, Any]: + return { + "components": [ + { + "component_id": "component-1", + "members": ["parent"], + "recommended": "deterministic", + "options": {"deterministic": {"capable": True}, "agentic": {"has_simplification": True}}, + }, + { + "component_id": "component-2", + "members": ["child"], + "recommended": "agentic", + "options": {"deterministic": {"capable": False}, "agentic": {"has_simplification": False}}, + }, + ], + "findings": [], + "default_plan": {"components": []}, + } + + +def _lfc_pipeline() -> dict[str, Any]: + """One agent-authored Lakeflow Connect pipeline via the AgenticComponentActivity escape hatch.""" + pipeline_definition = { + "name": "orders_ingestion", + "catalog": "${var.catalog}", + "target": "${var.schema}", + "ingestion_definition": {"connection_name": "c", "objects": []}, + } + return { + "name": "orders_lfc", + "tags": {"source": "adf"}, + "tasks": [ + { + "name": "Ingest orders", + "task_key": "ingest_orders", + "type": "AgenticComponentActivity", + "files": [], + "resources": [{"resource_key": "orders_ingestion", "definition": pipeline_definition}], + "task": {"pipeline_task": {"pipeline_id": "${resources.pipelines.orders_ingestion.id}"}}, + } + ], + } + + +def _named_lfc_pipeline(name: str) -> dict[str, Any]: + """A correctly-tagged authored combine pipeline with a caller-chosen ``name``.""" + pipeline = _lfc_pipeline() + pipeline["name"] = name + return pipeline + + +def _write_work(output_dir: Path, report: dict[str, Any], gaps: list[dict[str, Any]] | None = None) -> None: + work = output_dir / WORK_DIRNAME + work.mkdir(parents=True, exist_ok=True) + (work / REPORT_FILENAME).write_text(json.dumps(report, indent=2), encoding="utf-8") + if gaps is not None: + (work / GAPS_FILENAME).write_text(json.dumps(gaps, indent=2), encoding="utf-8") + + +def _node(task_key: str, native_type: str) -> SourceNode: + return SourceNode( + source_id=task_key, + task_key=task_key, + concept=CONCEPT_NOTEBOOK, + source="unit", + name=task_key, + native_type=native_type, + properties={STRATEGY_PROPERTY: "deterministic"}, + raw={"name": task_key, "type": native_type}, + ) + + +def _routed_inventory() -> dict[str, Any]: + """Inventory whose single 'parent -> child' control edge forms one component {child, parent}.""" + parent = SourceGraph( + name="parent", + source="unit", + tasks=[_node("call_child", "ExecutePipeline")], + lineage=Lineage( + control_edges=[ControlEdge(source_workflow="parent", target_workflow="child", via_task_key="call_child")] + ), + ) + child = SourceGraph(name="child", source="unit", tasks=[_node("copy_orders", "Copy")]) + return build_source_inventory([parent, child], source="unit", source_dir="/tmp/src") + + +def _setup_routed_agentic(output_dir: Path, *, decision: str = "agentic") -> None: + """Write inventory + report + a recorded conversion_plan.json routing {child, parent} per ``decision``.""" + _write_work(output_dir, _report_two_pipelines()) + metadata = output_dir / "metadata" + metadata.mkdir(parents=True, exist_ok=True) + (metadata / "inventory.json").write_text(json.dumps(_routed_inventory(), indent=2), encoding="utf-8") + plan = {"components": [{"component_id": "component-1", "members": ["child", "parent"], "decision": decision}]} + result = record_plan(output_dir, plan=plan) + assert result["ok"], result + + +# --------------------------------------------------------------------------- # +# Decision selection. +# --------------------------------------------------------------------------- # + + +def test_agentic_pipeline_names_selects_only_agentic_component_members() -> None: + plan = { + "components": [ + {"component_id": "component-1", "members": ["parent", "shared"], "decision": "agentic"}, + {"component_id": "component-2", "members": ["child"], "decision": "deterministic"}, + ] + } + assert agentic_pipeline_names(plan) == {"parent", "shared"} + + +def test_prompt_for_decisions_defaults_to_recommendation_and_honours_overrides() -> None: + answers = iter(["", "d"]) # component-1: accept (deterministic); component-2: override to deterministic + lines: list[str] = [] + plan = prompt_for_decisions(_recommendation(), input_fn=lambda _prompt: next(answers), output_fn=lines.append) + decisions = {c["component_id"]: c["decision"] for c in plan["components"]} + assert decisions == {"component-1": "deterministic", "component-2": "deterministic"} + # Each component carries its members so the plan validates against the inventory on record. + assert plan["components"][0]["members"] == ["parent"] + # The user saw the members and the prominent simplification option before answering. + assert any("component-1" in line for line in lines) + assert any("simplification" in line.lower() for line in lines) + + +def test_prompt_for_decisions_selects_agentic_on_a() -> None: + plan = prompt_for_decisions(_recommendation(), input_fn=lambda _prompt: "a", output_fn=lambda _line: None) + assert [c["decision"] for c in plan["components"]] == ["agentic", "agentic"] + + +# --------------------------------------------------------------------------- # +# Report alteration (the edit step). +# --------------------------------------------------------------------------- # + + +def test_alter_report_with_no_agentic_pipelines_is_identity() -> None: + report = _report_two_pipelines() + gaps: list[dict[str, Any]] = [] + original = copy.deepcopy(report) + new_report, new_gaps = alter_report(report, gaps, set()) + assert new_report == original + assert new_gaps == [] + + +def test_alter_report_placeholders_agentic_pipeline_and_leaves_deterministic_untouched() -> None: + report = _report_two_pipelines() + child_before = copy.deepcopy(report["pipelines"][1]) + new_report, new_gaps = alter_report(report, [], {"parent"}) + + parent = next(p for p in new_report["pipelines"] if p["name"] == "parent") + child = next(p for p in new_report["pipelines"] if p["name"] == "child") + # The agentic pipeline's deterministic Copy task became a placeholder marking the gap. + assert [task["type"] for task in parent["tasks"]] == ["PlaceholderActivity"] + placeholder = parent["tasks"][0] + assert placeholder["name"] == "Extract" and placeholder["task_key"] == "extract" + assert placeholder["original_type"] == "CopyActivity" + # The deterministic pipeline is byte-identical. + assert child == child_before + + +def test_alter_report_emits_one_gap_per_removed_task_tagged_with_pipeline() -> None: + _report, gaps = alter_report(_report_two_pipelines(), [], {"parent"}) + assert len(gaps) == 1 + gap = gaps[0] + assert gap["activity_name"] == "Extract" + assert gap["activity_type"] == "CopyActivity" + assert gap["pipeline"] == "parent" + + +def test_alter_report_emits_exactly_one_gap_per_task_replacing_prior_gaps() -> None: + report = { + "pipelines": [ + { + "name": "parent", + "tasks": [_copy_task("Extract", "extract"), _copy_task("Flow", "flow")], + } + ] + } + # A pre-existing (untagged) convert gap for a task that will be routed must not be duplicated. + existing_gaps = [{"activity_name": "Flow", "activity_type": "ExecuteDataFlow", "raw_definition": None}] + _report, gaps = alter_report(report, existing_gaps, {"parent"}) + # One gap per routed task, no duplicates, all tagged with the pipeline. + assert len(gaps) == 2 + identities = sorted((gap["pipeline"], gap["activity_name"]) for gap in gaps) + assert identities == [("parent", "Extract"), ("parent", "Flow")] + assert len(identities) == len(set(identities)) + + +def test_alter_report_is_idempotent_on_gaps() -> None: + report = {"pipelines": [{"name": "parent", "tasks": [_copy_task("Extract", "extract")]}]} + once_report, once_gaps = alter_report(report, [], {"parent"}) + twice_report, twice_gaps = alter_report(once_report, once_gaps, {"parent"}) + # Running the edit again does not append a second gap for the same routed task. + assert twice_gaps == once_gaps + assert len(twice_gaps) == 1 + + +def test_alter_report_keeps_a_non_routed_gap_that_shares_a_task_name_with_a_routed_task() -> None: + # Both pipelines have a task named "Sync"; only 'parent' is routed agentic. The untagged convert + # gap belongs to the non-routed 'other' and must survive -- the drop must not match by global name. + report = { + "pipelines": [ + {"name": "parent", "tasks": [_copy_task("Sync", "sync_parent")]}, + {"name": "other", "tasks": [_copy_task("Sync", "sync_other")]}, + ] + } + existing = [{"activity_name": "Sync", "activity_type": "ExecuteDataFlow", "raw_definition": None}] + _report, gaps = alter_report(report, existing, {"parent"}) + + # The non-routed pipeline's untagged "Sync" gap survives. + assert existing[0] in gaps + # The routed pipeline contributes exactly one tagged gap for its own "Sync" task. + tagged = [gap for gap in gaps if gap.get("pipeline") == "parent"] + assert tagged == [ + { + "activity_name": "Sync", + "activity_type": "CopyActivity", + "raw_definition": tagged[0]["raw_definition"], + "pipeline": "parent", + } + ] + # No duplicate tagged gap for the routed task. + assert len(tagged) == 1 + + +def test_alter_report_keeps_gaps_for_non_routed_pipelines() -> None: + report = { + "pipelines": [ + {"name": "parent", "tasks": [_copy_task("Extract", "extract")]}, + {"name": "other", "tasks": [_copy_task("Keep", "keep")]}, + ] + } + existing = [{"activity_name": "OtherGap", "activity_type": "Custom", "raw_definition": None, "pipeline": "other"}] + _report, gaps = alter_report(report, existing, {"parent"}) + # The non-routed pipeline's gap survives; the routed pipeline contributes exactly one. + assert {gap["activity_name"] for gap in gaps} == {"OtherGap", "Extract"} + + +def test_alter_report_preserves_depends_on_edges_on_placeholders() -> None: + report = { + "pipelines": [ + { + "name": "parent", + "tasks": [ + _copy_task("A", "a"), + {**_copy_task("B", "b"), "depends_on": [{"task_key": "a", "outcome": "Succeeded"}]}, + ], + } + ] + } + new_report, _gaps = alter_report(report, [], {"parent"}) + task_b = new_report["pipelines"][0]["tasks"][1] + assert task_b["type"] == "PlaceholderActivity" + assert task_b["depends_on"] == [{"task_key": "a", "outcome": "Succeeded"}] + + +def test_alter_report_handles_the_single_pipeline_report_shape() -> None: + report = {"name": "parent", "tasks": [_copy_task("Extract", "extract")]} + new_report, gaps = alter_report(report, [], {"parent"}) + assert new_report["tasks"][0]["type"] == "PlaceholderActivity" + assert len(gaps) == 1 + + +# --------------------------------------------------------------------------- # +# apply_plan_to_report: I/O + non-breaking golden. +# --------------------------------------------------------------------------- # + + +def test_apply_plan_to_report_edits_report_and_gaps_files(tmp_path: Path) -> None: + _write_work(tmp_path, _report_two_pipelines(), gaps=[]) + summary = apply_plan_to_report(tmp_path, _plan(parent="agentic", child="deterministic")) + assert summary["agentic_pipelines"] == ["parent"] + + report = json.loads((tmp_path / WORK_DIRNAME / REPORT_FILENAME).read_text(encoding="utf-8")) + parent = next(p for p in report["pipelines"] if p["name"] == "parent") + assert parent["tasks"][0]["type"] == "PlaceholderActivity" + gaps = json.loads((tmp_path / WORK_DIRNAME / GAPS_FILENAME).read_text(encoding="utf-8")) + assert [gap["pipeline"] for gap in gaps] == ["parent"] + + +def test_apply_plan_to_report_all_deterministic_leaves_files_byte_identical(tmp_path: Path) -> None: + _write_work(tmp_path, _report_two_pipelines(), gaps=[]) + report_path = tmp_path / WORK_DIRNAME / REPORT_FILENAME + gaps_path = tmp_path / WORK_DIRNAME / GAPS_FILENAME + report_before = report_path.read_bytes() + gaps_before = gaps_path.read_bytes() + + summary = apply_plan_to_report(tmp_path, _plan(parent="deterministic", child="deterministic")) + assert summary["agentic_pipelines"] == [] + assert report_path.read_bytes() == report_before + assert gaps_path.read_bytes() == gaps_before + + +# --------------------------------------------------------------------------- # +# Per-pipeline agentic fill: reuse the existing name-matched merge. +# --------------------------------------------------------------------------- # + + +def test_per_pipeline_fill_reuses_merge_agentic_results(tmp_path: Path) -> None: + _write_work(tmp_path, _report_two_pipelines(), gaps=[]) + apply_plan_to_report(tmp_path, _plan(parent="agentic", child="deterministic")) + report_path = tmp_path / WORK_DIRNAME / REPORT_FILENAME + + results_dir = tmp_path / "agentic_results" + results_dir.mkdir() + (results_dir / "extract.json").write_text( + json.dumps( + { + "pipeline": "parent", + "activity_name": "Extract", + "task": { + "type": "NotebookActivity", + "name": "Extract", + "task_key": "extract", + "notebook_path": "/Workspace/Shared/agentic_extract", + }, + } + ), + encoding="utf-8", + ) + merged, unmatched = merge_agentic_results(report_path, results_dir) + assert (merged, unmatched) == (1, 0) + + report = json.loads(report_path.read_text(encoding="utf-8")) + parent = next(p for p in report["pipelines"] if p["name"] == "parent") + assert parent["tasks"][0]["type"] == "NotebookActivity" + assert validate_report_structurally(report).ok + + +# --------------------------------------------------------------------------- # +# Cross-pipeline COMBINE: the new pipeline-grain fill (N pipelines -> one LFC). +# --------------------------------------------------------------------------- # + + +def test_combine_group_fill_replaces_group_pipelines_with_authored_ones() -> None: + report = _report_two_pipelines() + merged = combine_group_fill(report, {"parent", "child"}, [_lfc_pipeline()]) + names = [pipeline["name"] for pipeline in merged["pipelines"]] + assert names == ["orders_lfc"] + + +def test_combine_group_fill_keeps_pipelines_outside_the_group() -> None: + report = { + "pipelines": [ + {"name": "parent", "tasks": [_copy_task("Extract", "extract")]}, + {"name": "child", "tasks": [_notebook_task("Load", "load")]}, + {"name": "unrelated", "tasks": [_notebook_task("Keep", "keep")]}, + ] + } + merged = combine_group_fill(report, {"parent", "child"}, [_lfc_pipeline()]) + names = sorted(pipeline["name"] for pipeline in merged["pipelines"]) + assert names == ["orders_lfc", "unrelated"] + + +def test_combine_fill_of_two_pipelines_into_one_lfc_passes_structural_validation() -> None: + report = _report_two_pipelines() + merged = combine_group_fill(report, {"parent", "child"}, [_lfc_pipeline()]) + result = validate_report_structurally(merged) + assert result.ok, [f"{finding.code}: {finding.message}" for finding in result.findings] + + +def test_combine_fill_with_a_dangling_pipeline_reference_is_caught() -> None: + dangling = _lfc_pipeline() + # Point the pipeline_task at a resource that is never declared in this activity's resources list. + dangling["tasks"][0]["task"] = {"pipeline_task": {"pipeline_id": "${resources.pipelines.ghost.id}"}} + merged = combine_group_fill(_report_two_pipelines(), {"parent", "child"}, [dangling]) + result = validate_report_structurally(merged) + assert not result.ok + assert any(finding.code == "dangling_pipeline_reference" for finding in result.violations) + + +def test_apply_combine_fill_writes_merged_report_when_valid(tmp_path: Path) -> None: + _setup_routed_agentic(tmp_path, decision="agentic") + result = apply_combine_fill(tmp_path, ["parent", "child"], [_lfc_pipeline()]) + assert result["ok"] is True + assert result["component_id"] == "component-1" + report = json.loads((tmp_path / WORK_DIRNAME / REPORT_FILENAME).read_text(encoding="utf-8")) + assert [pipeline["name"] for pipeline in report["pipelines"]] == ["orders_lfc"] + + +def test_apply_combine_fill_rejects_and_does_not_write_on_dangling_reference(tmp_path: Path) -> None: + _setup_routed_agentic(tmp_path, decision="agentic") + report_before = (tmp_path / WORK_DIRNAME / REPORT_FILENAME).read_bytes() + dangling = _lfc_pipeline() + dangling["tasks"][0]["task"] = {"pipeline_task": {"pipeline_id": "${resources.pipelines.ghost.id}"}} + + result = apply_combine_fill(tmp_path, ["parent", "child"], [dangling]) + assert result["ok"] is False + assert any("dangling_pipeline_reference" in violation for violation in result["violations"]) + # The report on disk is untouched when validation fails. + assert (tmp_path / WORK_DIRNAME / REPORT_FILENAME).read_bytes() == report_before + + +def test_combine_rejects_an_authored_pipeline_missing_the_source_tag(tmp_path: Path) -> None: + """FIX 2: an authored pipeline without tags.source == 'adf' fails closed at combine (nothing written).""" + _setup_routed_agentic(tmp_path, decision="agentic") + report_before = (tmp_path / WORK_DIRNAME / REPORT_FILENAME).read_bytes() + untagged = _lfc_pipeline() + del untagged["tags"] + + result = apply_combine_fill(tmp_path, ["parent", "child"], [untagged]) + + assert result["ok"] is False + assert any("tags.source" in violation and "adf" in violation for violation in result["violations"]) + # Nothing is written when the source tag is missing. + assert (tmp_path / WORK_DIRNAME / REPORT_FILENAME).read_bytes() == report_before + + +def test_combine_rejects_an_authored_pipeline_with_wrong_source_tag(tmp_path: Path) -> None: + """A non-'adf' source tag is rejected the same way (routing/agentic is ADF-only).""" + _setup_routed_agentic(tmp_path, decision="agentic") + mistagged = _lfc_pipeline() + mistagged["tags"] = {"source": "airflow"} + + result = apply_combine_fill(tmp_path, ["parent", "child"], [mistagged]) + + assert result["ok"] is False + assert any("tags.source" in violation for violation in result["violations"]) + + +def test_combine_is_idempotent_running_twice_yields_no_duplicate(tmp_path: Path) -> None: + """FIX 3: re-running combine with the same members + authored pipeline does not duplicate it.""" + _setup_routed_agentic(tmp_path, decision="agentic") + + first = apply_combine_fill(tmp_path, ["parent", "child"], [_lfc_pipeline()]) + assert first["ok"] is True + assert first["already_combined"] is False + + # The recorded plan still lists {parent, child}, so the plan/membership check passes again; the + # report, however, now contains only the authored pipeline. A naive re-run would re-append it. + second = apply_combine_fill(tmp_path, ["parent", "child"], [_lfc_pipeline()]) + assert second["ok"] is True + assert second["already_combined"] is True + + report = json.loads((tmp_path / WORK_DIRNAME / REPORT_FILENAME).read_text(encoding="utf-8")) + names = [pipeline["name"] for pipeline in report["pipelines"]] + assert names == ["orders_lfc"] # exactly one authored pipeline, no duplicate + + +def test_combine_idempotent_with_a_differently_named_authored_pipeline(tmp_path: Path) -> None: + """FIX 3 (a): a second combine whose authored replacement is renamed must NOT duplicate. + + Name-set matching would fail here (the new name is absent from the report, the old members are + already gone), fall through, and append a second authored pipeline. Provenance keyed on + (component_id, fingerprint) catches the re-run regardless of the authored name. + """ + _setup_routed_agentic(tmp_path, decision="agentic") + + first = apply_combine_fill(tmp_path, ["parent", "child"], [_named_lfc_pipeline("orders_lfc")]) + assert first["ok"] is True and first["already_combined"] is False + + # Same members, but the authored replacement is named differently this time. + second = apply_combine_fill(tmp_path, ["parent", "child"], [_named_lfc_pipeline("orders_lfc_v2")]) + assert second["ok"] is True + assert second["already_combined"] is True + + report = json.loads((tmp_path / WORK_DIRNAME / REPORT_FILENAME).read_text(encoding="utf-8")) + names = [pipeline["name"] for pipeline in report["pipelines"]] + assert names == ["orders_lfc"] # first authored pipeline kept; the renamed re-run added nothing + + +def test_combine_idempotent_when_authored_name_collides_with_a_former_member(tmp_path: Path) -> None: + """FIX 3 (b): an authored name colliding with a former member is still detected as already-combined. + + After the first combine the report holds a pipeline named 'parent' (a former member). Name-based + detection would see 'parent' present and conclude the combine had not happened; provenance keeps + the detection correct and independent of names. + """ + _setup_routed_agentic(tmp_path, decision="agentic") + + first = apply_combine_fill(tmp_path, ["parent", "child"], [_named_lfc_pipeline("parent")]) + assert first["ok"] is True and first["already_combined"] is False + + second = apply_combine_fill(tmp_path, ["parent", "child"], [_named_lfc_pipeline("parent")]) + assert second["ok"] is True + assert second["already_combined"] is True + + report = json.loads((tmp_path / WORK_DIRNAME / REPORT_FILENAME).read_text(encoding="utf-8")) + names = [pipeline["name"] for pipeline in report["pipelines"]] + assert names == ["parent"] # exactly one authored pipeline, no duplicate + + +# --------------------------------------------------------------------------- # +# Combine membership is bound to the recorded, fingerprint-bound plan. +# --------------------------------------------------------------------------- # + + +def test_combine_rejects_a_deterministic_component(tmp_path: Path) -> None: + _setup_routed_agentic(tmp_path, decision="deterministic") + result = apply_combine_fill(tmp_path, ["parent", "child"], [_lfc_pipeline()]) + assert result["ok"] is False + assert "not agentic" in result["error"] + # A deterministic component is never swapped out. + report = json.loads((tmp_path / WORK_DIRNAME / REPORT_FILENAME).read_text(encoding="utf-8")) + assert sorted(pipeline["name"] for pipeline in report["pipelines"]) == ["child", "parent"] + + +def test_combine_rejects_a_partial_member_set(tmp_path: Path) -> None: + _setup_routed_agentic(tmp_path, decision="agentic") + result = apply_combine_fill(tmp_path, ["parent"], [_lfc_pipeline()]) + assert result["ok"] is False + assert "do not exactly match" in result["error"] + + +def test_combine_rejects_a_superset_member_set(tmp_path: Path) -> None: + _setup_routed_agentic(tmp_path, decision="agentic") + result = apply_combine_fill(tmp_path, ["parent", "child", "extra"], [_lfc_pipeline()]) + assert result["ok"] is False + assert "do not exactly match" in result["error"] + + +def test_combine_rejects_a_typoed_member(tmp_path: Path) -> None: + _setup_routed_agentic(tmp_path, decision="agentic") + result = apply_combine_fill(tmp_path, ["parent", "chld"], [_lfc_pipeline()]) + assert result["ok"] is False + assert "do not exactly match" in result["error"] + + +def test_combine_rejects_a_stale_fingerprint_plan(tmp_path: Path) -> None: + _setup_routed_agentic(tmp_path, decision="agentic") + # Mutate the inventory after recording so the recorded fingerprint no longer matches. + inventory_path = tmp_path / "metadata" / "inventory.json" + inventory = json.loads(inventory_path.read_text(encoding="utf-8")) + inventory["pipelines"].append({"name": "late_addition", "activities": [], "motifs": []}) + inventory_path.write_text(json.dumps(inventory, indent=2), encoding="utf-8") + + result = apply_combine_fill(tmp_path, ["parent", "child"], [_lfc_pipeline()]) + assert result["ok"] is False + assert "stale" in result["error"] + + +def test_combine_requires_a_recorded_plan(tmp_path: Path) -> None: + _write_work(tmp_path, _report_two_pipelines()) # report only; no plan recorded + result = apply_combine_fill(tmp_path, ["parent", "child"], [_lfc_pipeline()]) + assert result["ok"] is False + assert "conversion_plan.json" in result["error"] + + +def test_apply_combine_fill_has_no_validation_bypass() -> None: + # There must be no surface that writes the combined report without structural validation. + assert "validate" not in inspect.signature(apply_combine_fill).parameters diff --git a/tests/unit/test_route_audit.py b/tests/unit/test_route_audit.py new file mode 100644 index 0000000..09074d3 --- /dev/null +++ b/tests/unit/test_route_audit.py @@ -0,0 +1,91 @@ +"""Tests for the routing/gaps audit artifact the package phase persists (FIX 5). + +Package prunes the transient ``.work/`` folder by default, which erases the routing trail +(translation report + ``gaps.json``). ``_write_route_audit`` summarises the recorded plan's routed +components + decisions + the gaps routing introduced into ``metadata/route_audit.json`` so the trail +survives the prune. It must no-op (write nothing) when no plan was recorded, keeping the no-route +path byte-identical. +""" + +from __future__ import annotations + +import json +from pathlib import Path + +from flowx.bundler.dab_writer import _write_route_audit + + +def _write_plan(output_dir: Path) -> None: + metadata = output_dir / "metadata" + metadata.mkdir(parents=True, exist_ok=True) + plan = { + "schema_version": "1", + "inventory_sha256": "abc123", + "components": [ + {"component_id": "component-1", "members": ["parent", "child"], "decision": "agentic"}, + {"component_id": "component-2", "members": ["solo"], "decision": "deterministic"}, + ], + } + (metadata / "conversion_plan.json").write_text(json.dumps(plan, indent=2), encoding="utf-8") + + +def _write_gaps(output_dir: Path) -> None: + work = output_dir / ".work" + work.mkdir(parents=True, exist_ok=True) + gaps = [ + { + "activity_name": "Extract", + "activity_type": "CopyActivity", + "raw_definition": {"name": "Extract", "type": "CopyActivity"}, + "pipeline": "parent", + }, + { + "activity_name": "Load", + "activity_type": "NotebookActivity", + "raw_definition": {"name": "Load"}, + "pipeline": "child", + }, + ] + (work / "gaps.json").write_text(json.dumps(gaps, indent=2), encoding="utf-8") + + +def test_route_audit_summarises_components_decisions_and_gaps(tmp_path: Path) -> None: + _write_plan(tmp_path) + _write_gaps(tmp_path) + + audit_path = _write_route_audit(tmp_path) + + assert audit_path == tmp_path / "metadata" / "route_audit.json" + audit = json.loads(audit_path.read_text(encoding="utf-8")) + assert audit["recorded_against_inventory_sha256"] == "abc123" + # Both routed components and their decisions are recorded. + decisions = {component["component_id"]: component["decision"] for component in audit["components"]} + assert decisions == {"component-1": "agentic", "component-2": "deterministic"} + # Only the agentic component's members are surfaced as agentic pipelines. + assert audit["agentic_pipelines"] == ["child", "parent"] + # The gaps introduced survive as a compact summary (no verbose raw_definition). + assert audit["gaps_count"] == 2 + assert {gap["pipeline"] for gap in audit["gaps_introduced"]} == {"parent", "child"} + assert all("raw_definition" not in gap for gap in audit["gaps_introduced"]) + + +def test_route_audit_noops_without_a_recorded_plan(tmp_path: Path) -> None: + # No conversion_plan.json -> no routing happened -> nothing is written (no-route path stays clean). + (tmp_path / "metadata").mkdir(parents=True, exist_ok=True) + + audit_path = _write_route_audit(tmp_path) + + assert audit_path is None + assert not (tmp_path / "metadata" / "route_audit.json").exists() + + +def test_route_audit_handles_missing_gaps_file(tmp_path: Path) -> None: + # A recorded plan but no gaps.json (e.g. all-deterministic route) still writes an audit with zero gaps. + _write_plan(tmp_path) + + audit_path = _write_route_audit(tmp_path) + + assert audit_path is not None + audit = json.loads(audit_path.read_text(encoding="utf-8")) + assert audit["gaps_count"] == 0 + assert audit["gaps_introduced"] == [] diff --git a/tests/unit/test_routing_components.py b/tests/unit/test_routing_components.py new file mode 100644 index 0000000..d5b458e --- /dev/null +++ b/tests/unit/test_routing_components.py @@ -0,0 +1,152 @@ +"""Tests for connected-component computation over control lineage (routing, #77). + +The inventory fixtures are built by hand through the source-agnostic emitter +(:func:`flowx.discovery_inventory.build_source_inventory`) -- no ADF, no Airflow -- so the +routing engine is proven against the standardised inventory shape and its per-pipeline +control-edge lineage, exactly as it will see it in production. Components are weak/undirected: +two pipelines land in the same component when a resolved control edge joins them in either +direction. +""" + +from __future__ import annotations + +from typing import Any + +from flowx.discovery_inventory import STRATEGY_PROPERTY, build_source_inventory +from flowx.models.discovery import CONCEPT_NOTEBOOK, SourceGraph, SourceNode +from flowx.models.ir import ControlEdge, Lineage +from flowx.routing import build_components + + +def _node(task_key: str, native_type: str = "Notebook", *, strategy: str = "deterministic") -> SourceNode: + return SourceNode( + source_id=task_key, + task_key=task_key, + concept=CONCEPT_NOTEBOOK, + source="unit", + name=task_key, + native_type=native_type, + properties={STRATEGY_PROPERTY: strategy}, + raw={"name": task_key, "type": native_type}, + ) + + +def _graph(name: str, *, edges: list[ControlEdge] | None = None, nodes: list[SourceNode] | None = None) -> SourceGraph: + """A one-node graph named *name*, optionally carrying control edges to other graphs.""" + return SourceGraph( + name=name, + source="unit", + tasks=nodes if nodes is not None else [_node(f"{name}_task")], + lineage=Lineage(control_edges=edges or []), + ) + + +def _edge(source: str, target: str, via: str, *, resolved: bool = True) -> ControlEdge: + return ControlEdge(source_workflow=source, target_workflow=target, via_task_key=via, resolved=resolved) + + +def _inventory(graphs: list[SourceGraph]) -> dict[str, Any]: + return build_source_inventory(graphs, source="unit", source_dir="/tmp/src") + + +def test_isolated_pipelines_each_form_their_own_component() -> None: + inventory = _inventory([_graph("a"), _graph("b"), _graph("c")]) + components, findings = build_components(inventory) + assert components == [["a"], ["b"], ["c"]] + assert findings == [] + + +def test_a_chain_of_calls_forms_one_component() -> None: + graphs = [ + _graph("a", edges=[_edge("a", "b", "call_b")], nodes=[_node("call_b", "ExecutePipeline")]), + _graph("b", edges=[_edge("b", "c", "call_c")], nodes=[_node("call_c", "ExecutePipeline")]), + _graph("c"), + ] + components, findings = build_components(_inventory(graphs)) + assert components == [["a", "b", "c"]] + assert findings == [] + + +def test_a_branch_fans_into_one_component() -> None: + graphs = [ + _graph( + "a", + edges=[_edge("a", "b", "call_b"), _edge("a", "c", "call_c")], + nodes=[_node("call_b", "ExecutePipeline"), _node("call_c", "ExecutePipeline")], + ), + _graph("b"), + _graph("c"), + ] + components, _ = build_components(_inventory(graphs)) + assert components == [["a", "b", "c"]] + + +def test_a_cycle_forms_one_component() -> None: + graphs = [ + _graph("a", edges=[_edge("a", "b", "call_b")], nodes=[_node("call_b", "ExecutePipeline")]), + _graph("b", edges=[_edge("b", "a", "call_a")], nodes=[_node("call_a", "ExecutePipeline")]), + ] + components, findings = build_components(_inventory(graphs)) + assert components == [["a", "b"]] + assert findings == [] + + +def test_multiple_edges_between_the_same_pair_still_one_component() -> None: + graphs = [ + _graph( + "a", + edges=[_edge("a", "b", "call_b1"), _edge("a", "b", "call_b2")], + nodes=[_node("call_b1", "ExecutePipeline"), _node("call_b2", "ExecutePipeline")], + ), + _graph("b"), + ] + components, _ = build_components(_inventory(graphs)) + assert components == [["a", "b"]] + + +def test_two_separate_clusters_are_distinct_components() -> None: + graphs = [ + _graph("a", edges=[_edge("a", "b", "call_b")], nodes=[_node("call_b", "ExecutePipeline")]), + _graph("b"), + _graph("x", edges=[_edge("x", "y", "call_y")], nodes=[_node("call_y", "ExecutePipeline")]), + _graph("y"), + ] + components, _ = build_components(_inventory(graphs)) + assert components == [["a", "b"], ["x", "y"]] + + +def test_unresolved_callee_is_a_finding_not_a_severed_edge() -> None: + graphs = [ + _graph( + "a", edges=[_edge("a", "", "call_ghost", resolved=False)], nodes=[_node("call_ghost", "ExecutePipeline")] + ) + ] + components, findings = build_components(_inventory(graphs)) + # 'a' still forms its own component; the unresolved edge is recorded, not dropped. + assert components == [["a"]] + assert len(findings) == 1 + assert "call_ghost" in findings[0] and "unresolved" in findings[0].lower() + + +def test_edge_to_pipeline_absent_from_inventory_is_a_finding() -> None: + graphs = [ + _graph("a", edges=[_edge("a", "ghost", "call_ghost")], nodes=[_node("call_ghost", "ExecutePipeline")]), + _graph("b"), + ] + components, findings = build_components(_inventory(graphs)) + assert components == [["a"], ["b"]] + assert len(findings) == 1 + assert "ghost" in findings[0] + + +def test_component_membership_and_ordering_are_deterministic() -> None: + # Declared out of alphabetical order; members and components must still sort deterministically. + graphs = [ + _graph("z"), + _graph("m", edges=[_edge("m", "a", "call_a")], nodes=[_node("call_a", "ExecutePipeline")]), + _graph("a"), + ] + first, _ = build_components(_inventory(graphs)) + second, _ = build_components(_inventory(graphs)) + assert first == second + assert first == [["a", "m"], ["z"]] diff --git a/tests/unit/test_routing_recommend.py b/tests/unit/test_routing_recommend.py new file mode 100644 index 0000000..a272538 --- /dev/null +++ b/tests/unit/test_routing_recommend.py @@ -0,0 +1,331 @@ +"""Tests for the per-component conversion recommendation (routing, #77). + +Each component surfaces BOTH options as first-class peers: a deterministic option (engine-capability +assessment + motif/coverage evidence + uncovered gaps) and an agentic option (recommended Databricks +patterns from the insights block, with any simplification pattern surfaced prominently). The +``recommended`` field is a library-computed starting suggestion -- deterministic only when the whole +component is engine-capable. +""" + +from __future__ import annotations + +from typing import Any + +from flowx.discovery_inventory import STRATEGY_PROPERTY, build_source_inventory +from flowx.models.discovery import CONCEPT_NOTEBOOK, SourceGraph, SourceNode +from flowx.models.ir import ControlEdge, Lineage +from flowx.models.motifs import MOTIF_METADATA_DRIVEN_BULK_COPY, DetectedMotif +from flowx.routing import build_recommendation, recommend_component + + +def _node(task_key: str, native_type: str, *, strategy: str = "deterministic") -> SourceNode: + return SourceNode( + source_id=task_key, + task_key=task_key, + concept=CONCEPT_NOTEBOOK, + source="unit", + name=task_key, + native_type=native_type, + properties={STRATEGY_PROPERTY: strategy}, + raw={"name": task_key, "type": native_type}, + ) + + +def _inventory( + graphs: list[SourceGraph], + *, + motifs: dict[str, list[DetectedMotif]] | None = None, + insights: dict[str, Any] | None = None, +) -> dict[str, Any]: + inventory = build_source_inventory(graphs, source="unit", source_dir="/tmp/src", motifs_by_pipeline=motifs) + if insights is not None: + inventory["insights"] = insights + return inventory + + +def test_all_deterministic_component_recommends_deterministic() -> None: + graphs = [SourceGraph(name="a", source="unit", tasks=[_node("load", "Notebook"), _node("copy", "Copy")])] + recommended, options = recommend_component(["a"], _inventory(graphs)) + assert recommended == "deterministic" + assert options["deterministic"]["capable"] is True + assert options["deterministic"]["uncovered"] == [] + assert options["deterministic"]["activity_counts"] == {"deterministic": 2, "agentic": 0, "unsupported": 0} + + +def test_uncovered_agentic_activity_recommends_agentic_and_cites_the_gap() -> None: + graphs = [ + SourceGraph( + name="a", + source="unit", + tasks=[_node("load", "Notebook"), _node("flow", "ExecuteDataFlow", strategy="agentic")], + ) + ] + recommended, options = recommend_component(["a"], _inventory(graphs)) + assert recommended == "agentic" + assert options["deterministic"]["capable"] is False + assert options["deterministic"]["uncovered"] == [ + {"pipeline": "a", "activity": "flow", "type": "ExecuteDataFlow", "strategy": "agentic"} + ] + assert options["deterministic"]["activity_counts"] == {"deterministic": 1, "agentic": 1, "unsupported": 0} + + +def test_motif_covered_agentic_activity_is_engine_capable() -> None: + # An agentic-strategy activity that a detected motif claims is covered -> the component stays + # engine-capable and the recommendation is deterministic. + graphs = [ + SourceGraph( + name="a", + source="unit", + tasks=[_node("lookup", "Lookup"), _node("each", "ForEach", strategy="agentic")], + ) + ] + motif = DetectedMotif( + definition=MOTIF_METADATA_DRIVEN_BULK_COPY, + matched_activities=["lookup", "each"], + source_type_hint="database", + ) + recommended, options = recommend_component(["a"], _inventory(graphs, motifs={"a": [motif]})) + assert recommended == "deterministic" + assert options["deterministic"]["capable"] is True + assert options["deterministic"]["uncovered"] == [] + assert options["deterministic"]["motifs"] == ["metadata_driven_bulk_copy"] + + +def test_motif_coverage_does_not_leak_across_pipelines_in_a_component() -> None: + # A component spans A -> B. Pipeline A has a motif claiming an activity named 'shared'; pipeline B + # has an UNRELATED activity also named 'shared' with no motif. Motif coverage must be keyed by + # (pipeline, activity), so B's 'shared' stays an uncovered gap rather than being masked by A's. + graphs = [ + SourceGraph( + name="A", + source="unit", + tasks=[_node("call_b", "ExecutePipeline"), _node("shared", "Copy", strategy="agentic")], + lineage=Lineage( + control_edges=[ControlEdge(source_workflow="A", target_workflow="B", via_task_key="call_b")] + ), + ), + SourceGraph(name="B", source="unit", tasks=[_node("shared", "Copy", strategy="agentic")]), + ] + motif = DetectedMotif( + definition=MOTIF_METADATA_DRIVEN_BULK_COPY, matched_activities=["shared"], source_type_hint="database" + ) + recommended, options = recommend_component(["A", "B"], _inventory(graphs, motifs={"A": [motif]})) + assert recommended == "agentic" + assert options["deterministic"]["capable"] is False + # A's 'shared' is motif-covered; only B's unrelated 'shared' is the uncovered gap. + assert options["deterministic"]["uncovered"] == [ + {"pipeline": "B", "activity": "shared", "type": "Copy", "strategy": "agentic"} + ] + + +def test_missing_strategy_counts_as_unsupported_gap() -> None: + node = _node("mystery", "Custom") + node.properties.pop(STRATEGY_PROPERTY) # no strategy recorded at all + graphs = [SourceGraph(name="a", source="unit", tasks=[node])] + recommended, options = recommend_component(["a"], _inventory(graphs)) + assert recommended == "agentic" + assert options["deterministic"]["activity_counts"] == {"deterministic": 0, "agentic": 0, "unsupported": 1} + assert options["deterministic"]["uncovered"][0]["activity"] == "mystery" + + +def test_agentic_option_surfaces_insight_patterns_with_simplification_prominent() -> None: + graphs = [SourceGraph(name="a", source="unit", tasks=[_node("load", "Notebook")])] + insights = { + "pipeline_insights": [ + { + "pipeline": "a", + "recommended_patterns": [ + { + "pattern": "Lakeflow Connect SQL Server connector", + "fit": "Replaces the bespoke extractor", + "simplification_pattern": True, + }, + {"pattern": "Parameterised Lakeflow Job", "fit": "Like-for-like", "simplification_pattern": False}, + ], + } + ] + } + _, options = recommend_component(["a"], _inventory(graphs, insights=insights)) + agentic = options["agentic"] + assert agentic["has_simplification"] is True + assert agentic["recommended_patterns"] == [ + { + "pipeline": "a", + "pattern": "Lakeflow Connect SQL Server connector", + "fit": "Replaces the bespoke extractor", + "simplification_pattern": True, + }, + { + "pipeline": "a", + "pattern": "Parameterised Lakeflow Job", + "fit": "Like-for-like", + "simplification_pattern": False, + }, + ] + + +def test_agentic_option_empty_when_no_insights() -> None: + graphs = [SourceGraph(name="a", source="unit", tasks=[_node("load", "Notebook")])] + _, options = recommend_component(["a"], _inventory(graphs)) + assert options["agentic"] == { + "recommended_patterns": [], + "has_simplification": False, + "release_disclosures": [], + } + + +# --------------------------------------------------------------------------- # +# Agentic option: neutral GA/Preview release-state disclosure (no warnings). +# --------------------------------------------------------------------------- # + + +def _insights_one_pattern(pattern: dict[str, Any]) -> dict[str, Any]: + """An insights block with a single recommended pattern on pipeline 'a'.""" + return {"pipeline_insights": [{"pipeline": "a", "recommended_patterns": [pattern]}]} + + +def test_agentic_option_ga_and_unknown_are_silent() -> None: + # 'ga' and 'unknown' both contribute NO disclosure entry -- 'unknown' is treated exactly like 'ga'. + for state in ("ga", "unknown"): + graphs = [SourceGraph(name="a", source="unit", tasks=[_node("load", "Notebook")])] + insights = _insights_one_pattern( + { + "pattern": "Lakeflow Job", + "fit": "Native orchestration", + "simplification_pattern": False, + "release_state": state, + } + ) + _, options = recommend_component(["a"], _inventory(graphs, insights=insights)) + assert options["agentic"]["release_disclosures"] == [], f"{state!r} must be silent" + + +def test_agentic_option_public_preview_is_disclosed_as_production_ready() -> None: + graphs = [SourceGraph(name="a", source="unit", tasks=[_node("load", "Notebook")])] + insights = _insights_one_pattern( + { + "pattern": "Lakeflow Connect", + "fit": "Managed ingestion", + "simplification_pattern": True, + "release_state": "public_preview", + "release_state_source": "https://docs.databricks.com/ingestion/lakeflow-connect/", + } + ) + _, options = recommend_component(["a"], _inventory(graphs, insights=insights)) + disclosures = options["agentic"]["release_disclosures"] + assert len(disclosures) == 1 + disclosure = disclosures[0] + assert disclosure["release_state"] == "public_preview" + assert disclosure["label"] == "Public Preview (production-ready)" + assert disclosure["pipeline"] == "a" + # A neutral disclosure -- no warning/severity framing. + assert "severity" not in disclosure + # The cited source rides along into the human-readable disclosure. + assert "docs.databricks.com" in disclosure["message"] + + +def test_agentic_option_private_preview_and_beta_are_plain_labels() -> None: + for state, expected_label in (("private_preview", "Private Preview"), ("beta", "Beta")): + graphs = [SourceGraph(name="a", source="unit", tasks=[_node("load", "Notebook")])] + insights = _insights_one_pattern( + { + "pattern": "Some capability", + "fit": "A capability", + "simplification_pattern": False, + "release_state": state, + "release_state_source": "https://docs.databricks.com/some-feature/", + } + ) + _, options = recommend_component(["a"], _inventory(graphs, insights=insights)) + disclosures = options["agentic"]["release_disclosures"] + assert len(disclosures) == 1 + assert disclosures[0]["label"] == expected_label + assert disclosures[0]["release_state"] == state + # Plain factual label -- no warning/severity field. + assert "severity" not in disclosures[0] + + +def test_agentic_option_discloses_each_non_silent_pattern() -> None: + # A component with a public-preview pattern and a private-preview pattern discloses BOTH as plain + # neutral labels (public preview noted production-ready); a ga/unknown pattern would add nothing. + graphs = [ + SourceGraph( + name="parent", + source="unit", + tasks=[_node("call_child", "ExecutePipeline")], + lineage=Lineage( + control_edges=[ + ControlEdge(source_workflow="parent", target_workflow="child", via_task_key="call_child") + ] + ), + ), + SourceGraph(name="child", source="unit", tasks=[_node("load", "Notebook")]), + ] + insights = { + "pipeline_insights": [ + { + "pipeline": "parent", + "recommended_patterns": [ + { + "pattern": "Public thing", + "fit": "x", + "simplification_pattern": False, + "release_state": "public_preview", + "release_state_source": "https://docs.databricks.com/a/", + } + ], + }, + { + "pipeline": "child", + "recommended_patterns": [ + { + "pattern": "Gated thing", + "fit": "y", + "simplification_pattern": False, + "release_state": "private_preview", + "release_state_source": "https://docs.databricks.com/b/", + } + ], + }, + ] + } + _, options = recommend_component(["child", "parent"], _inventory(graphs, insights=insights)) + disclosures = options["agentic"]["release_disclosures"] + labels = {d["release_state"]: d["label"] for d in disclosures} + assert labels == {"public_preview": "Public Preview (production-ready)", "private_preview": "Private Preview"} + + +def test_build_recommendation_emits_both_options_and_a_default_plan() -> None: + graphs = [ + SourceGraph( + name="parent", + source="unit", + tasks=[_node("call_child", "ExecutePipeline")], + lineage=Lineage( + control_edges=[ + ControlEdge(source_workflow="parent", target_workflow="child", via_task_key="call_child") + ] + ), + ), + SourceGraph(name="child", source="unit", tasks=[_node("flow", "ExecuteDataFlow", strategy="agentic")]), + SourceGraph(name="solo", source="unit", tasks=[_node("load", "Notebook")]), + ] + recommendation = build_recommendation(_inventory(graphs)) + ids = [component["component_id"] for component in recommendation["components"]] + assert ids == ["component-1", "component-2"] + first = recommendation["components"][0] + assert first["members"] == ["child", "parent"] + assert set(first["options"]) == {"deterministic", "agentic"} + assert first["recommended"] == "agentic" # child's ExecuteDataFlow is an uncovered gap + assert recommendation["components"][1]["recommended"] == "deterministic" # solo notebook + # The default plan proposes decision == recommended for every component, ready to record as-is. + assert recommendation["default_plan"]["components"] == [ + {"component_id": "component-1", "members": ["child", "parent"], "decision": "agentic"}, + {"component_id": "component-2", "members": ["solo"], "decision": "deterministic"}, + ] + + +def test_recommendation_is_pure_and_idempotent() -> None: + graphs = [SourceGraph(name="a", source="unit", tasks=[_node("load", "Notebook")])] + inventory = _inventory(graphs) + assert build_recommendation(inventory) == build_recommendation(inventory) diff --git a/tests/unit/test_routing_record.py b/tests/unit/test_routing_record.py new file mode 100644 index 0000000..c8ea44f --- /dev/null +++ b/tests/unit/test_routing_record.py @@ -0,0 +1,388 @@ +"""Tests for validating an authored conversion plan and recording it (routing, #77). + +The agent authors ONLY the per-component decision (and an optional rationale); the library recomputes +members, recommended, and both options' evidence on record, and binds the plan to the inventory with +a SHA-256 fingerprint. The recorded ``conversion_plan.json`` is a separate additive artifact -- it +never mutates ``inventory.json``. Fixtures are built through the source-agnostic emitter so the engine +is proven against the production inventory shape. +""" + +from __future__ import annotations + +import copy +import json +from pathlib import Path +from typing import Any + +import pytest + +from flowx.adapter.__main__ import main as adapter_cli_main +from flowx.discovery_insights import inventory_fingerprint +from flowx.discovery_inventory import STRATEGY_PROPERTY, build_source_inventory +from flowx.models.conversion_plan import ( + SCHEMA_VERSION, + ComponentPlan, + ConversionPlan, +) +from flowx.models.discovery import CONCEPT_NOTEBOOK, CONCEPT_RUN_WORKFLOW, SourceGraph, SourceNode +from flowx.models.ir import ControlEdge, Lineage +from flowx.routing import ( + PLAN_FILENAME, + build_recommendation, + load_plan, + record_plan, + validate_plan, +) + + +def _node(task_key: str, native_type: str, *, strategy: str = "deterministic") -> SourceNode: + return SourceNode( + source_id=task_key, + task_key=task_key, + concept=CONCEPT_NOTEBOOK, + source="unit", + name=task_key, + native_type=native_type, + properties={STRATEGY_PROPERTY: strategy}, + raw={"name": task_key, "type": native_type}, + ) + + +def _inventory() -> dict[str, Any]: + """A two-pipeline component (parent -> child, both engine-capable) plus a standalone 'solo'. + + 'child' carries a Lakeflow Connect simplification insight, so its component is deterministic-capable + AND has a prominent agentic re-architecture option -- the deterministic-vs-LFC choice #77 surfaces. + """ + parent = SourceGraph( + name="parent", + source="unit", + tasks=[_node("call_child", "ExecutePipeline")], + lineage=Lineage( + control_edges=[ControlEdge(source_workflow="parent", target_workflow="child", via_task_key="call_child")] + ), + ) + child = SourceGraph(name="child", source="unit", tasks=[_node("copy_orders", "Copy")]) + solo = SourceGraph(name="solo", source="unit", tasks=[_node("load", "Notebook")]) + inventory = build_source_inventory([parent, child, solo], source="unit", source_dir="/tmp/src") + inventory["insights"] = { + "pipeline_insights": [ + { + "pipeline": "child", + "recommended_patterns": [ + { + "pattern": "Lakeflow Connect SQL Server connector", + "fit": "Replaces the bespoke Copy extractor with a managed pipeline", + "simplification_pattern": True, + } + ], + } + ] + } + return inventory + + +def _authored_plan() -> dict[str, Any]: + """Accept-the-recommendation plan: decision == recommended for both components.""" + return { + "components": [ + {"component_id": "component-1", "members": ["child", "parent"], "decision": "deterministic"}, + {"component_id": "component-2", "members": ["solo"], "decision": "deterministic"}, + ] + } + + +def _write_inventory(output_dir: Path, inventory: dict[str, Any]) -> Path: + metadata = output_dir / "metadata" + metadata.mkdir(parents=True, exist_ok=True) + path = metadata / "inventory.json" + path.write_text(json.dumps(inventory, indent=2), encoding="utf-8") + return path + + +# --------------------------------------------------------------------------- # +# Models. +# --------------------------------------------------------------------------- # + + +def test_models_construct_and_document_the_contract() -> None: + plan = ConversionPlan( + inventory_sha256="abc", + components=[ + ComponentPlan( + component_id="component-1", + members=["child", "parent"], + recommended="deterministic", + decision="agentic", + ) + ], + ) + assert plan.schema_version == SCHEMA_VERSION + assert plan.components[0].decision == "agentic" + + +# --------------------------------------------------------------------------- # +# Validation: success (including the default plan round-trip). +# --------------------------------------------------------------------------- # + + +def test_valid_plan_passes_validation() -> None: + assert validate_plan(_authored_plan(), _inventory()) == [] + + +def test_default_plan_from_recommendation_validates() -> None: + inventory = _inventory() + default_plan = build_recommendation(inventory)["default_plan"] + assert validate_plan(default_plan, inventory) == [] + + +# --------------------------------------------------------------------------- # +# Validation: failure modes. +# --------------------------------------------------------------------------- # + + +def test_non_dict_payload_is_a_violation() -> None: + assert validate_plan([1, 2], _inventory()) == ["conversion plan must be a JSON object, got list"] + + +def test_unknown_top_level_key_including_library_owned_fields() -> None: + raw = _authored_plan() + raw["schema_version"] = "1" + raw["bogus"] = True + violations = validate_plan(raw, _inventory()) + assert any("unknown top-level key: 'schema_version' (set by the library, not the author)" in v for v in violations) + assert any("unknown top-level key: 'bogus'" in v for v in violations) + + +def test_library_owned_component_fields_are_rejected() -> None: + raw = _authored_plan() + raw["components"][0]["recommended"] = "deterministic" + raw["components"][0]["options"] = {} + violations = validate_plan(raw, _inventory()) + assert any("unknown field 'recommended' (set by the library, not the author)" in v for v in violations) + assert any("unknown field 'options' (set by the library, not the author)" in v for v in violations) + + +def test_invalid_decision_value_is_rejected() -> None: + raw = _authored_plan() + raw["components"][0]["decision"] = "maybe" + violations = validate_plan(raw, _inventory()) + assert any("'decision' must be one of" in v for v in violations) + + +def test_unknown_pipeline_in_members_is_rejected() -> None: + raw = _authored_plan() + raw["components"][0]["members"] = ["child", "ghost"] + violations = validate_plan(raw, _inventory()) + assert any("pipeline 'ghost' not in inventory" in v for v in violations) + + +def test_members_that_do_not_form_a_component_are_rejected() -> None: + # 'parent' and 'solo' are in different components; pairing them is not a coherent route. + raw = { + "components": [ + {"component_id": "component-1", "members": ["parent", "solo"], "decision": "deterministic"}, + ] + } + violations = validate_plan(raw, _inventory()) + assert any("do not form a connected component" in v for v in violations) + + +def test_partial_plan_missing_a_component_is_rejected() -> None: + raw = { + "components": [ + {"component_id": "component-1", "members": ["child", "parent"], "decision": "deterministic"}, + ] + } + violations = validate_plan(raw, _inventory()) + assert any("every component must be routed" in v for v in violations) + + +def test_duplicate_decision_for_one_component_is_rejected() -> None: + raw = { + "components": [ + {"component_id": "component-1", "members": ["child", "parent"], "decision": "deterministic"}, + {"component_id": "component-1", "members": ["child", "parent"], "decision": "agentic"}, + {"component_id": "component-2", "members": ["solo"], "decision": "deterministic"}, + ] + } + violations = validate_plan(raw, _inventory()) + assert any("decided" in v and "times" in v for v in violations) + + +def test_component_id_mismatch_is_rejected() -> None: + raw = _authored_plan() + raw["components"][0]["component_id"] = "component-9" + violations = validate_plan(raw, _inventory()) + assert any("does not match" in v for v in violations) + + +def test_duplicate_members_are_rejected() -> None: + # A frozenset would collapse ["solo", "solo"] to {"solo"} and wrongly match component-2; the + # members-match/bijection contract must reject the duplicate explicitly. + raw = _authored_plan() + raw["components"][1]["members"] = ["solo", "solo"] + violations = validate_plan(raw, _inventory()) + assert any("duplicate" in v.lower() and "solo" in v for v in violations) + + +def test_rationale_must_be_non_empty_when_present() -> None: + raw = _authored_plan() + raw["components"][0]["rationale"] = " " + violations = validate_plan(raw, _inventory()) + assert any("'rationale' must be a non-empty string" in v for v in violations) + + +# --------------------------------------------------------------------------- # +# Record: fingerprint binding, both-options shape, atomicity, idempotency. +# --------------------------------------------------------------------------- # + + +def test_record_writes_plan_with_both_options_recommended_and_fingerprint(tmp_path: Path) -> None: + inventory = _inventory() + _write_inventory(tmp_path, inventory) + result = record_plan(tmp_path, plan=_authored_plan()) + assert result["ok"] is True + assert result["components"] == 2 + + plan = json.loads((tmp_path / "metadata" / PLAN_FILENAME).read_text(encoding="utf-8")) + assert plan["schema_version"] == SCHEMA_VERSION + assert plan["inventory_sha256"] == inventory_fingerprint(inventory) + component = plan["components"][0] + assert component["members"] == ["child", "parent"] + assert component["recommended"] == "deterministic" + assert component["decision"] == "deterministic" + assert set(component["options"]) == {"deterministic", "agentic"} + # The agentic option's simplification pattern is surfaced as a first-class peer, not buried. + assert component["options"]["agentic"]["has_simplification"] is True + assert component["options"]["deterministic"]["capable"] is True + + +def test_record_preserves_a_user_override_with_both_recommended_and_decision(tmp_path: Path) -> None: + inventory = _inventory() + _write_inventory(tmp_path, inventory) + raw = _authored_plan() + # User overrides the deterministic recommendation to take the Lakeflow Connect re-architecture. + raw["components"][0]["decision"] = "agentic" + raw["components"][0]["rationale"] = "Adopt the managed connector to retire the extractor" + record_plan(tmp_path, plan=raw) + plan = json.loads((tmp_path / "metadata" / PLAN_FILENAME).read_text(encoding="utf-8")) + component = plan["components"][0] + assert component["recommended"] == "deterministic" + assert component["decision"] == "agentic" + assert component["rationale"] == "Adopt the managed connector to retire the extractor" + + +def test_record_leaves_inventory_byte_identical(tmp_path: Path) -> None: + inventory = _inventory() + inventory_path = _write_inventory(tmp_path, inventory) + original_bytes = inventory_path.read_bytes() + record_plan(tmp_path, plan=_authored_plan()) + assert inventory_path.read_bytes() == original_bytes + + +def test_record_is_idempotent(tmp_path: Path) -> None: + inventory = _inventory() + _write_inventory(tmp_path, inventory) + plan_path = tmp_path / "metadata" / PLAN_FILENAME + record_plan(tmp_path, plan=_authored_plan()) + first = plan_path.read_bytes() + record_plan(tmp_path, plan=_authored_plan()) + assert plan_path.read_bytes() == first + + +def test_record_leaves_plan_untouched_on_validation_failure(tmp_path: Path) -> None: + inventory = _inventory() + _write_inventory(tmp_path, inventory) + record_plan(tmp_path, plan=_authored_plan()) # write a good plan first + good_bytes = (tmp_path / "metadata" / PLAN_FILENAME).read_bytes() + + bad = _authored_plan() + bad["components"][0]["decision"] = "nonsense" + result = record_plan(tmp_path, plan=bad) + assert result["ok"] is False and result["violations"] + assert (tmp_path / "metadata" / PLAN_FILENAME).read_bytes() == good_bytes + + +def test_record_raises_when_inventory_missing(tmp_path: Path) -> None: + with pytest.raises(FileNotFoundError): + record_plan(tmp_path, plan=_authored_plan()) + + +def test_load_plan_requires_exactly_one_source(tmp_path: Path) -> None: + with pytest.raises(ValueError): + load_plan() + with pytest.raises(ValueError): + load_plan(plan={}, plan_path=tmp_path / "x.json") + + +# --------------------------------------------------------------------------- # +# CLI wiring. The reshaped `route` is one command: no plan on a non-TTY emits the recommendation; a +# `--plan-path` records the plan and edits the report. (Report editing is covered end-to-end in +# test_cli_route_agentic.py; here we exercise the record/recommend wiring against a bare inventory.) +# --------------------------------------------------------------------------- # + + +def _write_empty_report(output_dir: Path) -> None: + """A minimal report matching the inventory's pipelines so the edit step has something to rewrite.""" + work = output_dir / ".work" + work.mkdir(parents=True, exist_ok=True) + report = { + "pipelines": [ + {"name": "parent", "tasks": [{"name": "call_child", "task_key": "call_child", "type": "CopyActivity"}]}, + {"name": "child", "tasks": [{"name": "copy_orders", "task_key": "copy_orders", "type": "CopyActivity"}]}, + {"name": "solo", "tasks": [{"name": "load", "task_key": "load", "type": "NotebookActivity"}]}, + ] + } + (work / "translation_report.json").write_text(json.dumps(report, indent=2), encoding="utf-8") + + +def test_cli_route_without_plan_emits_components( + tmp_path: Path, capsys: pytest.CaptureFixture[str], monkeypatch: pytest.MonkeyPatch +) -> None: + import io + + _write_inventory(tmp_path, _inventory()) + monkeypatch.setattr("sys.stdin", io.StringIO("")) # not a TTY -> dry-run recommendation + code = adapter_cli_main(["route", "--output-dir", str(tmp_path)]) + assert code == 0 + payload = json.loads(capsys.readouterr().out) + assert [c["component_id"] for c in payload["components"]] == ["component-1", "component-2"] + assert "default_plan" in payload + + +def test_cli_route_with_plan_records_and_reports_the_edit(tmp_path: Path, capsys: pytest.CaptureFixture[str]) -> None: + _write_inventory(tmp_path, _inventory()) + _write_empty_report(tmp_path) + plan_path = tmp_path / "plan.json" + plan_path.write_text(json.dumps(_authored_plan()), encoding="utf-8") + code = adapter_cli_main(["route", "--output-dir", str(tmp_path), "--plan-path", str(plan_path)]) + assert code == 0 + payload = json.loads(capsys.readouterr().out) + assert payload["ok"] is True and payload["components"] == 2 + assert "edit" in payload + assert (tmp_path / "metadata" / PLAN_FILENAME).exists() + + +def test_cli_route_validation_failure_returns_1(tmp_path: Path, capsys: pytest.CaptureFixture[str]) -> None: + _write_inventory(tmp_path, _inventory()) + _write_empty_report(tmp_path) + raw = _authored_plan() + raw["components"].pop() # partial plan: component-2 undecided + plan_path = tmp_path / "plan.json" + plan_path.write_text(json.dumps(raw), encoding="utf-8") + code = adapter_cli_main(["route", "--output-dir", str(tmp_path), "--plan-path", str(plan_path)]) + assert code == 1 + payload = json.loads(capsys.readouterr().out) + assert payload["ok"] is False and payload["violations"] + + +def test_fixture_isolation() -> None: + a = _authored_plan() + b = _authored_plan() + a["components"][0]["decision"] = "agentic" + assert b["components"][0]["decision"] == "deterministic" + assert copy.deepcopy(a) == a + # CONCEPT_RUN_WORKFLOW is imported for symmetry with the discovery fixtures; touch it so linters + # see the dependency used. + assert isinstance(CONCEPT_RUN_WORKFLOW, str)