From eb768bb898cd80e2f2861da2b557c3f176eee2a2 Mon Sep 17 00:00:00 2001 From: Matthew Moorcroft Date: Wed, 16 Sep 2026 17:24:33 +0100 Subject: [PATCH 01/25] ADF discovery: inventory + lineage + motifs on the shared source AST Map ADF exports onto the shared source-neutral discovery AST and emit a source-agnostic inventory, populate control/data lineage (Switch-aware), surface detected ADF motifs additively in the inventory, and carry the lineage/motif tags on the IR (data_reads/data_writes, motif_id on the Activity base; pipeline lineage block) with matching serialization. Consolidates PRs #79 (ADF -> shared source AST + inventory emitter), #80 (data/control lineage + Switch-aware walking), #81 (emit lineage into inventory.json) and #90/#64 (motif detection + exact-duplicate dedupe), plus the coverage golden. Co-authored-by: Isaac --- src/flowx/bundler/dab_writer.py | 66 +- src/flowx/discovery_inventory.py | 245 ++++ src/flowx/ir_serde.py | 10 +- src/flowx/lineage.py | 217 ++- src/flowx/models/ir.py | 17 +- src/flowx/sources/adf/dataset_lineage.py | 495 +++++++ src/flowx/sources/adf/discovery_mapping.py | 394 ++++++ src/flowx/sources/adf/loader.py | 111 +- .../golden/adf_fixture_coverage.json | 1162 +++++++++++++++++ tests/unit/test_adf_dataset_lineage.py | 239 ++++ tests/unit/test_adf_discovery_mapping.py | 423 ++++++ tests/unit/test_adf_inventory_superset.py | 193 +++ tests/unit/test_discovery_inventory.py | 361 +++++ tests/unit/test_lineage_substrate.py | 454 +++++++ 14 files changed, 4305 insertions(+), 82 deletions(-) create mode 100644 src/flowx/discovery_inventory.py create mode 100644 src/flowx/sources/adf/dataset_lineage.py create mode 100644 src/flowx/sources/adf/discovery_mapping.py create mode 100644 tests/resources/golden/adf_fixture_coverage.json create mode 100644 tests/unit/test_adf_dataset_lineage.py create mode 100644 tests/unit/test_adf_discovery_mapping.py create mode 100644 tests/unit/test_adf_inventory_superset.py create mode 100644 tests/unit/test_discovery_inventory.py create mode 100644 tests/unit/test_lineage_substrate.py diff --git a/src/flowx/bundler/dab_writer.py b/src/flowx/bundler/dab_writer.py index 436900da..8cdf20b2 100644 --- a/src/flowx/bundler/dab_writer.py +++ b/src/flowx/bundler/dab_writer.py @@ -30,7 +30,10 @@ from flowx.models.ir import ( Activity, AppendVariableActivity, + ControlEdge, CopyActivity, + DataAsset, + DataEdge, DbtFactoryActivity, DeleteActivity, Dependency, @@ -38,8 +41,10 @@ FilterActivity, ForEachActivity, IfConditionActivity, + Lineage, LookupActivity, MotifActivity, + MotifAnnotation, NotebookActivity, Pipeline, PlaceholderActivity, @@ -2037,6 +2042,7 @@ def pipeline_dict_to_ir(pipeline_dict: dict[str, Any]) -> tuple[Pipeline, list[d reconciliation_status=pipeline_dict.get("reconciliation_status"), migration_status=pipeline_dict.get("migration_status", "included"), audit=dict(pipeline_dict.get("audit") or {}), + lineage=_reconstruct_lineage(pipeline_dict.get("lineage")), ) return pipeline, parameters @@ -2258,9 +2264,10 @@ def _reconstruct_ir(task_ir: dict[str, Any]) -> Activity: bridge_required_parameters=dict(task_ir.get("bridge_required_parameters") or {}), ) if task_type == "MotifActivity": + # motif_id arrives via ``base`` (Activity now owns the field); passing it again here + # would raise "multiple values for keyword argument 'motif_id'". return MotifActivity( **base, - motif_id=task_ir.get("motif_id", "unknown"), display_name=task_ir.get("display_name", base["name"]), databricks_replacement=task_ir.get("databricks_replacement", "notebook"), matched_activity_names=list(task_ir.get("matched_activity_names", [])), @@ -2311,9 +2318,66 @@ def _common_activity_kwargs(task_ir: dict[str, Any]) -> dict[str, Any]: "required_parameters": dict(task_ir.get("required_parameters") or {}), "compute_mode": task_ir.get("compute_mode"), "notifications": task_ir.get("notifications"), + "motif_id": task_ir.get("motif_id"), + "data_reads": _reconstruct_data_assets(task_ir.get("data_reads")), + "data_writes": _reconstruct_data_assets(task_ir.get("data_writes")), } +def _reconstruct_data_assets(raw: list[dict[str, Any]] | None) -> list[DataAsset]: + """Rehydrates serialised DataAsset dicts into typed :class:`DataAsset` nodes.""" + if not raw: + return [] + return [ + DataAsset( + signature=asset.get("signature", ""), + identity=asset.get("identity"), + asset_type=asset.get("asset_type"), + properties=dict(asset.get("properties") or {}), + ) + for asset in raw + ] + + +def _reconstruct_lineage(raw: dict[str, Any] | None) -> Lineage | None: + """Rehydrates a serialised lineage block into a typed :class:`Lineage`, or ``None``.""" + if not raw: + return None + return Lineage( + control_edges=[ + ControlEdge( + source_workflow=edge.get("source_workflow", ""), + target_workflow=edge.get("target_workflow", ""), + via_task_key=edge.get("via_task_key", ""), + wait_for_completion=edge.get("wait_for_completion"), + resolved=bool(edge.get("resolved", True)), + ) + for edge in raw.get("control_edges") or [] + ], + data_edges=[ + DataEdge( + source_task_key=edge.get("source_task_key", ""), + target_task_key=edge.get("target_task_key", ""), + match_kind=edge.get("match_kind", ""), + match_key=edge.get("match_key", ""), + identity=edge.get("identity"), + asset_type=edge.get("asset_type"), + ) + for edge in raw.get("data_edges") or [] + ], + motifs=[ + MotifAnnotation( + motif_id=motif.get("motif_id", ""), + member_task_keys=list(motif.get("member_task_keys") or []), + display_name=motif.get("display_name"), + databricks_replacement=motif.get("databricks_replacement"), + notes=list(motif.get("notes") or []), + ) + for motif in raw.get("motifs") or [] + ], + ) + + def _reconstruct_dependencies(raw: list[dict[str, Any]] | None) -> list[Dependency] | None: if not raw: return None diff --git a/src/flowx/discovery_inventory.py b/src/flowx/discovery_inventory.py new file mode 100644 index 00000000..12a80c1a --- /dev/null +++ b/src/flowx/discovery_inventory.py @@ -0,0 +1,245 @@ +"""Source-agnostic projection of the shared discovery AST to ``inventory.json``. + +The discover phase writes ``metadata/inventory.json`` and the reporting layer +(:mod:`flowx.reporting.coverage`) and MCP surface (:mod:`flowx.mcp.runner`) read +it back. Historically each source built that JSON straight from its own AST, so +the shape drifted per source. This module is the single place that turns the +shared discovery AST (:mod:`flowx.models.discovery`) into the inventory shape, so +every source that maps onto :class:`~flowx.models.discovery.SourceGraph` emits the +*same* top-level document -- ``{source, source_dir, pipelines, summary}`` -- from +one code path. + +The projection is deliberately small and additive over the historical ADF shape: + +* top level gains a ``source`` discriminator (``"adf"`` / ``"airflow"``); +* each activity keeps its byte-compatible ``name`` / ``type`` / ``strategy`` (and + ``depends_on`` names when present) and gains the standardised + ``original_type``, ``dependencies`` (upstream **with conditions**), and the + verbatim per-node ``raw``; +* each pipeline entry gains an additive ``lineage`` block (control + data edges) + when its :class:`~flowx.models.discovery.SourceGraph` carries derived lineage; +* each pipeline entry gains an additive ``motifs`` list -- the multi-activity ADF + patterns the profiler *detects* at discover time (see + :mod:`flowx.motifs.detector`) surfaced verbatim, **without** collapsing the + member activities. Motifs are their own inventory concept and are deliberately + **decoupled from the lineage block** (they do not nest under ``lineage.motifs``, + which stays a convert-time IR concern); the key is omitted when a pipeline has + no detected motif. Collapse remains a *convert* decision + (:mod:`flowx.motifs.collapser`), never a discover one, so every member activity + still appears as its own entry in ``activities``; +* the ``summary`` keeps the historical count block. + +Lineage is placed per pipeline -- one block beside that pipeline's ``activities`` +-- to mirror the shared discovery serde, where lineage is a per-graph field +(:func:`flowx.discovery_serde.source_graph_to_dict`). The block is produced by the +one shared serialiser (:func:`flowx.ir_serde.lineage_to_dict`, the same one the +serde consumes), so an emitted block is byte-identical to the serde's and +round-trips through :func:`flowx.discovery_serde.source_graph_from_dict`. It is a +new key only: a graph with no derived lineage (``graph.lineage is None``) omits it +entirely, so the historical consumer keys (``source`` / ``pipelines`` / +``activities`` / ``summary``) are untouched. + +A node's translation ``strategy`` is a Databricks-*target* classification rather +than a source concept, so it is not a typed field on the discovery AST. By +convention a mapper stashes it under ``node.properties["strategy"]`` (see +:data:`STRATEGY_PROPERTY`); this module reads it there. Detected motifs are +handled the same way -- they carry a Databricks-*target* replacement and are not +a shared source concept, so they are not a typed field on the AST either; a +source supplies them per pipeline via ``motifs_by_pipeline`` and this module +projects them. Anything else a source wants to layer on -- Airflow's +audited-count block, findings, reconciliation status -- rides additively on top +of this base and is out of scope here. +""" + +from __future__ import annotations + +from collections.abc import Mapping +from typing import Any + +from flowx.ir_serde import lineage_to_dict +from flowx.models.discovery import ContainerNode, SourceGraph, SourceNode +from flowx.models.motifs import DetectedMotif + +# Well-known property key under which a mapper records a node's Databricks-target +# translation strategy ("deterministic" / "agentic" / "unsupported"). Kept in the +# free-form properties seam because strategy is a target concern, not a shared +# source concept, so it earns no typed field on the discovery AST. +STRATEGY_PROPERTY = "strategy" + +_DETERMINISTIC = "deterministic" +_AGENTIC = "agentic" + + +def build_source_inventory( + graphs: list[SourceGraph], + *, + source: str, + source_dir: str, + include_empty_pipelines: bool = True, + motifs_by_pipeline: Mapping[str, list[DetectedMotif]] | None = None, +) -> dict[str, Any]: + """Project a list of source graphs into the ``inventory.json`` document. + + Args: + graphs: The source workflows to inventory, already mapped onto the shared + discovery AST. + source: Source discriminator for the top-level ``source`` field + (``SOURCE_ADF`` / ``SOURCE_AIRFLOW``). + source_dir: Original source directory, echoed back for provenance. + include_empty_pipelines: When ``False``, a graph that contributes no + activities is left out of the ``pipelines`` list but still counted in + ``summary.pipeline_count`` -- this reproduces ADF's long-standing + behaviour of omitting zero-activity pipelines from the per-pipeline + listing while still reporting them in the totals. Sources that list + every workflow (Airflow) leave this ``True``. + motifs_by_pipeline: Optional map from pipeline name to the motifs a source + detected in it (see :mod:`flowx.motifs.detector`). A pipeline with a + non-empty entry gains an additive ``motifs`` list; the key is omitted + otherwise. This is surfacing only -- the member activities are never + collapsed here (collapse is a convert decision). Exact-duplicate + detections (same ``motif_id`` and same member set) are collapsed to one + (see :func:`_dedupe_motif_entries`); overlapping-but-distinct matches + are all kept. Keyed by + :attr:`~flowx.models.discovery.SourceGraph.name`, so a source with no + motif detector simply passes ``None``. + + Returns: + A JSON-friendly dict with ``source``, ``source_dir``, ``pipelines`` and + ``summary`` keys. + """ + motifs_by_pipeline = motifs_by_pipeline or {} + pipeline_entries: list[dict[str, Any]] = [] + deterministic = 0 + agentic = 0 + unsupported = 0 + + for graph in graphs: + flattened = _flatten_nodes(graph.tasks) + for node in flattened: + strategy = node.properties.get(STRATEGY_PROPERTY) + if strategy == _DETERMINISTIC: + deterministic += 1 + elif strategy == _AGENTIC: + agentic += 1 + else: + unsupported += 1 + if flattened or include_empty_pipelines: + entry: dict[str, Any] = { + "name": graph.name, + "activities": [_activity_entry(node) for node in flattened], + } + if graph.lineage is not None: + entry["lineage"] = lineage_to_dict(graph.lineage) + detected = motifs_by_pipeline.get(graph.name) + if detected: + entry["motifs"] = _dedupe_motif_entries([_motif_entry(motif) for motif in detected]) + pipeline_entries.append(entry) + + total = deterministic + agentic + unsupported + coverage_pct = round((deterministic + agentic) / total * 100, 1) if total else 0.0 + + return { + "source": source, + "source_dir": source_dir, + "pipelines": pipeline_entries, + "summary": { + "pipeline_count": len(graphs), + "activity_count": total, + "deterministic_count": deterministic, + "agentic_count": agentic, + "unsupported_count": unsupported, + "coverage_pct": coverage_pct, + }, + } + + +def _flatten_nodes(nodes: list[SourceNode]) -> list[SourceNode]: + """Flatten container branches into one depth-first activity list. + + Order is parent, then each branch's children in the branch's own insertion + order, recursively -- so an ``IfCondition``'s ``true`` branch precedes its + ``false`` branch and a ``Switch``'s cases precede its ``default``, matching the + order the source declared them. + """ + flattened: list[SourceNode] = [] + for node in nodes: + flattened.append(node) + if isinstance(node, ContainerNode): + for children in node.branches.values(): + flattened.extend(_flatten_nodes(children)) + return flattened + + +def _activity_entry(node: SourceNode) -> dict[str, Any]: + """Build one per-activity inventory entry from a node. + + The first three keys (plus ``depends_on`` when the node has dependencies) + reproduce the historical ADF activity shape byte-for-byte; the rest are the + additive standardised fields. + """ + entry: dict[str, Any] = { + "name": node.name if node.name is not None else node.task_key, + "type": node.native_type, + "strategy": node.properties.get(STRATEGY_PROPERTY), + } + upstream_names = [dependency.upstream for dependency in node.dependencies] + if upstream_names: + entry["depends_on"] = upstream_names + + entry["original_type"] = node.native_type + entry["dependencies"] = [ + { + "upstream": dependency.upstream, + "conditions": list(dependency.conditions), + "resolved": dependency.resolved, + } + for dependency in node.dependencies + ] + if node.raw is not None: + entry["raw"] = node.raw + return entry + + +def _motif_entry(motif: DetectedMotif) -> dict[str, Any]: + """Project one detected motif into its additive inventory entry. + + Surfacing only: ``member_task_keys`` names the participating activities as the + detector claimed them (for ADF these are the activity names, which are the + same values used as each activity entry's ``name`` / task key, so a consumer + can join a motif back to its members). The field names mirror the existing + :class:`~flowx.models.ir.MotifAnnotation` vocabulary so the two motif views + read the same, while staying a separate key from the ``lineage`` block. The + detector reports its confidence as human-readable rationale rather than a + numeric score, so ``confidence_notes`` carries that verbatim. + """ + return { + "motif_id": motif.definition.motif_id, + "display_name": motif.definition.display_name, + "databricks_replacement": motif.definition.databricks_replacement, + "member_task_keys": list(motif.matched_activities), + "source_type_hint": motif.source_type_hint, + "confidence_notes": list(motif.confidence_notes), + } + + +def _dedupe_motif_entries(entries: list[dict[str, Any]]) -> list[dict[str, Any]]: + """Collapse exact-duplicate motif entries, preserving first-seen order. + + A detector can report the same match more than once (e.g. two upstreams that + each pair with the same notification activity), which would otherwise emit + identical inventory entries. Two entries are the *same* motif only when they + share both ``motif_id`` **and** the exact same set of ``member_task_keys``; + such duplicates collapse to the first occurrence. Entries that merely overlap + -- same ``motif_id`` but a different member set -- are genuinely distinct + matches and are all kept. The member comparison is order-insensitive (a set), + so the same activities in a different order still count as one motif. + """ + seen: set[tuple[str, frozenset[str]]] = set() + deduped: list[dict[str, Any]] = [] + for entry in entries: + identity = (entry["motif_id"], frozenset(entry["member_task_keys"])) + if identity in seen: + continue + seen.add(identity) + deduped.append(entry) + return deduped diff --git a/src/flowx/ir_serde.py b/src/flowx/ir_serde.py index 6783e5b3..72b00a61 100644 --- a/src/flowx/ir_serde.py +++ b/src/flowx/ir_serde.py @@ -82,6 +82,8 @@ def pipeline_to_dict(pipeline: Pipeline) -> dict[str, Any]: } if pipeline.translation_configuration is not None: result["translation_configuration"] = configuration_to_dict(pipeline.translation_configuration) + if pipeline.lineage is not None: + result["lineage"] = lineage_to_dict(pipeline.lineage) return result @@ -216,6 +218,12 @@ def activity_to_dict(task: Activity) -> dict[str, Any]: task_dict["libraries"] = task.libraries if task.parameter_approximations: task_dict["parameter_approximations"] = task.parameter_approximations + if task.motif_id: + task_dict["motif_id"] = task.motif_id + if task.data_reads: + task_dict["data_reads"] = [data_asset_to_dict(asset) for asset in task.data_reads] + if task.data_writes: + task_dict["data_writes"] = [data_asset_to_dict(asset) for asset in task.data_writes] extra = activity_extra_fields(task) task_dict.update(extra) @@ -410,7 +418,7 @@ def activity_extra_fields(activity: Activity) -> dict[str, Any]: if activity.job_parameters: extra["job_parameters"] = activity.job_parameters case MotifActivity(): - extra["motif_id"] = activity.motif_id + # motif_id now lives on the Activity base and is serialised by activity_to_dict. extra["display_name"] = activity.display_name extra["databricks_replacement"] = activity.databricks_replacement extra["matched_activity_names"] = activity.matched_activity_names diff --git a/src/flowx/lineage.py b/src/flowx/lineage.py index 9e5d636a..447e265f 100644 --- a/src/flowx/lineage.py +++ b/src/flowx/lineage.py @@ -1,24 +1,76 @@ -"""Source-neutral lineage-edge cores. - -Primitive-level building blocks for deriving lineage edges. They take only -``task_key`` strings and :class:`~flowx.models.ir.DataAsset` values -- never a -``Pipeline`` or any ``Activity`` -- so the tier-matching, self-edge drop, and -dedup rules live in exactly one place. The discovery-AST derivation in -:mod:`flowx.discovery_lineage` gathers those primitives from a -:class:`~flowx.models.discovery.SourceGraph` and delegates here; the IR-facing -derivation that gathers them from a :class:`~flowx.models.ir.Pipeline` is added -separately in the convert->package lineage work so it stacks on this standard. - -They import nothing from ``sources/adf`` or ``sources/airflow``. Everything here -is pure: the functions read their inputs and return new edge lists; nothing is -mutated. +"""Source-neutral lineage derivation over the flowx Pipeline IR. + +These functions turn an already-translated :class:`~flowx.models.ir.Pipeline` +into its :class:`~flowx.models.ir.Lineage` block. They operate on IR primitives +only -- ``Activity`` subclasses, ``DataAsset``, ``task_key`` -- and import nothing +from ``sources/adf`` or ``sources/airflow`` so both front-ends share one code +path once they populate ``data_reads`` / ``data_writes`` / ``motif_id``. + +The tier-matching, self-edge drop, and dedup rules are factored into two +primitive-level cores -- :func:`control_edges_from_calls` and +:func:`data_edges_from_endpoints` -- that take only ``task_key`` strings and +:class:`DataAsset` values, never a ``Pipeline``. The IR entry points +(:func:`build_control_edges` / :func:`build_data_edges`) gather those primitives +from a pipeline and delegate, and the source-neutral discovery-AST derivation in +:mod:`flowx.discovery_lineage` gathers the same primitives from a +:class:`~flowx.models.discovery.SourceGraph` and delegates too, so both phases +join edges through exactly one implementation. + +Everything here is pure: the functions read the pipeline and return new edge +lists / a new :class:`Lineage`; nothing is mutated. :func:`with_lineage` attaches +a block by returning a *new* ``Pipeline`` rather than mutating the input, unlike +the in-place dependency rewrite in ``motifs/collapser.py``. """ from __future__ import annotations -from collections.abc import Iterable +import dataclasses +from collections.abc import Iterable, Iterator + +from flowx.models.ir import ( + Activity, + ControlEdge, + DataAsset, + DataEdge, + ExecutePipelineActivity, + ForEachActivity, + IfConditionActivity, + Lineage, + MotifActivity, + MotifAnnotation, + Pipeline, + RunJobActivity, + SwitchActivity, +) + + +def walk_activities(activities: list[Activity]) -> Iterator[Activity]: + """Yield every activity in *activities*, descending into control-flow containers. + + Recurses into ForEach inner activities, both If-condition branches, and every + Switch case plus its default branch, so a nested ExecutePipeline or a data + asset buried inside a Switch case is still reached. Motif ``original_activities`` + are intentionally not traversed: they are the pre-collapse originals kept for + reference, not live graph members. -from flowx.models.ir import ControlEdge, DataAsset, DataEdge + Args: + activities: Top-level (or already-nested) activity list to walk. + + Yields: + Each activity, container nodes included, in depth-first order. + """ + for activity in activities: + yield activity + match activity: + case ForEachActivity(): + yield from walk_activities(activity.inner_activities) + case IfConditionActivity(): + yield from walk_activities(activity.if_true_activities) + yield from walk_activities(activity.if_false_activities) + case SwitchActivity(): + for case_branch in activity.cases: + yield from walk_activities(case_branch.activities) + yield from walk_activities(activity.default_activities) def control_edges_from_calls( @@ -27,10 +79,10 @@ def control_edges_from_calls( ) -> list[ControlEdge]: """Assemble deduplicated control edges from raw invocation primitives. - The shared core behind the discovery-AST control-edge derivation (and the - IR-facing derivation added in the convert->package work): it owns the - self-edge drop, the dedup, and the unresolved-callee recording so those rules - live in exactly one place and every phase behaves identically. + The shared core behind :func:`build_control_edges` (IR) and the discovery-AST + control-edge derivation: it owns the self-edge drop, the dedup, and the + unresolved-callee recording so those rules live in exactly one place and both + phases behave identically. Args: source_workflow: Name of the calling workflow (pipeline / DAG). @@ -65,6 +117,35 @@ def control_edges_from_calls( return edges +def build_control_edges(pipeline: Pipeline) -> list[ControlEdge]: + """Derive cross-workflow invocation edges for a pipeline. + + Emits one :class:`ControlEdge` per invoking activity -- an + ``ExecutePipelineActivity`` (ADF) or a ``RunJobActivity`` (Airflow) -- found + anywhere in the pipeline, including inside ForEach / If / Switch containers + (fan-out is preserved: each call site is its own edge). Edges whose callee + equals the caller are dropped (no self-edges), and identical edges are + collapsed (no duplicates). An unresolved callee is recorded with + ``resolved=False`` rather than dropped. + + Args: + pipeline: The translated pipeline IR. + + Returns: + Deduplicated list of control edges, in first-seen order. + """ + + def _calls() -> Iterator[tuple[str, bool | None, str]]: + for activity in walk_activities(pipeline.tasks): + match activity: + case ExecutePipelineActivity(): + yield activity.pipeline_name or "", activity.wait_on_completion, activity.task_key + case RunJobActivity(): + yield activity.job_name or "", None, activity.task_key + + return control_edges_from_calls(pipeline.name, _calls()) + + def _match_assets(producer: DataAsset, consumer: DataAsset) -> tuple[str, str, str | None] | None: """Decide whether a written asset hands off to a read asset, and how. @@ -95,10 +176,9 @@ def data_edges_from_endpoints( ) -> list[DataEdge]: """Join producer endpoints to consumer endpoints via the two-tier match. - The shared core behind the discovery-AST data-edge derivation (and the - IR-facing derivation added in the convert->package work): it owns the - :func:`_match_assets` tier logic, the no-self-edge rule, and the dedup, so - every phase joins identically. + The shared core behind :func:`build_data_edges` (IR) and the discovery-AST + data-edge derivation: it owns the :func:`_match_assets` tier logic, the + no-self-edge rule, and the dedup, so both phases join identically. Args: producers: ``(task_key, written asset)`` pairs, in first-seen order. @@ -137,3 +217,92 @@ def data_edges_from_endpoints( ) ) return edges + + +def build_data_edges(pipeline: Pipeline) -> list[DataEdge]: + """Derive proven producer -> consumer data hand-offs for a pipeline. + + A producer is any activity with a ``data_writes`` asset; a consumer any + activity with a ``data_reads`` asset, gathered across the whole pipeline + (ForEach / If / Switch bodies included). Each producer asset is joined against + each consumer asset via :func:`_match_assets`, tagging the edge as an + ``identity`` or ``signature`` match. An activity never hands off to itself + (no self-edges), and identical edges are collapsed (no duplicates). + + Args: + pipeline: The translated pipeline IR. + + Returns: + Deduplicated list of data edges, in first-seen order. + """ + activities = list(walk_activities(pipeline.tasks)) + producers = [(activity.task_key, asset) for activity in activities for asset in activity.data_writes] + consumers = [(activity.task_key, asset) for activity in activities for asset in activity.data_reads] + return data_edges_from_endpoints(producers, consumers) + + +def build_motif_annotations(pipeline: Pipeline) -> list[MotifAnnotation]: + """Derive motif annotations from the collapsed motif activities in a pipeline. + + One annotation per :class:`MotifActivity`, listing the task keys it spans + (the motif task itself plus any member activities that carry the same + ``motif_id`` tag). Deduplicated by ``motif_id`` in first-seen order. + + Args: + pipeline: The translated pipeline IR. + + Returns: + List of motif annotations. + """ + annotations: list[MotifAnnotation] = [] + seen: set[str] = set() + activities = list(walk_activities(pipeline.tasks)) + for activity in activities: + if not isinstance(activity, MotifActivity): + continue + if activity.motif_id in seen: + continue + seen.add(activity.motif_id) + members = [activity.task_key] + members.extend( + other.task_key for other in activities if other is not activity and other.motif_id == activity.motif_id + ) + annotations.append( + MotifAnnotation( + motif_id=activity.motif_id, + member_task_keys=members, + display_name=activity.display_name, + databricks_replacement=activity.databricks_replacement, + notes=list(activity.confidence_notes), + ) + ) + return annotations + + +def build_lineage(pipeline: Pipeline) -> Lineage: + """Compose the full source-neutral lineage block for a pipeline. + + Args: + pipeline: The translated pipeline IR. + + Returns: + A :class:`Lineage` with control edges, data edges, and motif annotations. + """ + return Lineage( + control_edges=build_control_edges(pipeline), + data_edges=build_data_edges(pipeline), + motifs=build_motif_annotations(pipeline), + ) + + +def with_lineage(pipeline: Pipeline, lineage: Lineage) -> Pipeline: + """Return a *new* pipeline carrying *lineage*, leaving the input untouched. + + Args: + pipeline: The pipeline to copy. + lineage: The lineage block to attach. + + Returns: + A shallow copy of *pipeline* with ``lineage`` set. + """ + return dataclasses.replace(pipeline, lineage=lineage) diff --git a/src/flowx/models/ir.py b/src/flowx/models/ir.py index ffc7e456..a1b097a5 100644 --- a/src/flowx/models/ir.py +++ b/src/flowx/models/ir.py @@ -160,7 +160,8 @@ class MotifAnnotation: Source-neutral record tying a motif id to the tasks that belong to it, so the lineage block can report motifs without depending on how any particular - source detects them. ``member_task_keys`` lists the tasks the motif spans. + source detects them. ``member_task_keys`` lists the tasks the motif spans, + and each member activity also carries the same :attr:`Activity.motif_id` tag. Attributes: motif_id: Identifier of the matched motif definition. @@ -234,6 +235,13 @@ class Activity: compute_mode: str | None = None # Collapsed activity_and_notify spec set by the adapter: {destination, events, args, destination_name}. notifications: dict[str, Any] | None = None + # Lineage substrate (#61): source-neutral data assets this activity reads from and writes to, + # populated by per-source extractors in follow-up work. Always lists, never None. + data_reads: list[DataAsset] = field(default_factory=list) + data_writes: list[DataAsset] = field(default_factory=list) + # Id of the motif this activity was folded into (or belongs to); None when it is part of no motif. + # Owned here so every activity type -- not only MotifActivity -- can carry the tag. + motif_id: str | None = None @dataclass(slots=True, kw_only=True) @@ -710,6 +718,10 @@ class PlaceholderActivity(Activity): class MotifActivity(Activity): """Activity produced by collapsing a detected motif pattern. + Redeclares :attr:`Activity.motif_id` as required (the base owns the field so + every activity type can carry the tag and it round-trips through one code + path, but a motif activity always has one). + Attributes: motif_id: Identifier of the matched motif definition. display_name: Human-readable motif name. @@ -764,6 +776,8 @@ class Pipeline: reconciliation_status: Source-audit result for this pipeline. migration_status: Whether the pipeline is included or explicitly excluded. audit: Source-audit counts and transformation ledger. + lineage: Source-neutral lineage block (control/data edges + motif + annotations), or ``None`` when lineage has not been derived. """ name: str @@ -780,6 +794,7 @@ class Pipeline: audit: dict[str, Any] = field(default_factory=dict) translation_configuration: TranslationConfiguration | None = None bundle_variables: dict[str, dict[str, Any]] = field(default_factory=dict) + lineage: Lineage | None = None @dataclass(frozen=True, slots=True) diff --git a/src/flowx/sources/adf/dataset_lineage.py b/src/flowx/sources/adf/dataset_lineage.py new file mode 100644 index 00000000..ba51aa7b --- /dev/null +++ b/src/flowx/sources/adf/dataset_lineage.py @@ -0,0 +1,495 @@ +"""Resolve ADF dataset references into source-neutral :class:`DataAsset` values. + +This is the ADF half of data-lineage population (#62b). It re-homes the +dataset-identity resolver and the path-signature logic first written for the +closed #36 ADF-only lineage attempt, but emits the shared #61 two-tier +:class:`~flowx.models.ir.DataAsset` (``identity`` + ``signature``) instead of a +bespoke edge type, so the source-neutral join in :mod:`flowx.lineage` / +:mod:`flowx.discovery_lineage` does the matching. + +Two tiers, exactly as #36 established them: + +* **identity** -- the resolved physical location of the asset (``schema.table`` or + a concrete ``abfss://`` path). Present only when it resolves *deterministically* + from literals; a parameterised reference is never guessed at and leaves + ``identity`` unset. This is the strong join key. +* **signature** -- the *path-derived* weak key. For an asset with a resolved + identity the signature mirrors that identity, so the weak tier can never join a + resolved asset to an unresolved one on a coincidence. For an unresolved reference + the signature is the *structural path signature* (#36's "expression" tier: the + literal path skeleton plus its parameter-slot count) when the path has a literal + anchor. It is **never** the bare dataset reference name: two unrelated opaque + references that merely share a name must not join (#36's explicit rule), so when + neither a physical identity nor a path-anchored signature is available the + signature is left empty and the asset cannot participate in signature matching. + +Only literal, provable values ever become an ``identity`` -- the resolver returns +``None`` rather than guessing, which is what stopped #36's spurious edges. +""" + +from __future__ import annotations + +import json +import re +from collections.abc import Iterator +from typing import Any + +from flowx.models.adf_ast import AdfActivity, AdfDatasetReference, AdfDefinitions +from flowx.models.ir import DataAsset, TranslationContext +from flowx.parser.expression_parser import resolve_expression, resolve_interpolated_string + +_ACCOUNT_NAME_RE = re.compile(r"AccountName=([A-Za-z0-9]+)", re.IGNORECASE) +_DATASET_PARAM_RE = re.compile(r"^@dataset\(\)\.([A-Za-z_][A-Za-z0-9_]*)$") + +# Runtime references inside a path expression. Each is a value only knowable at +# run time; for a *structural* signature we collapse them all to one slot token so +# that, e.g., a writer's ``pipeline().parameters.entityID`` and a reader's +# ``item().entityID`` (the same value passed down a ForEach) share the same shape. +_PARAM_REF_RE = re.compile( + r"pipeline\(\)\.parameters\.\w+" + r"|item\(\)(?:\.\w+)*" + r"|variables\('[^']*'\)" + r"|dataset\(\)\.\w+" + r"|activity\('[^']*'\)\.[\w.]+" +) +_QUOTED_LITERAL_RE = re.compile(r"'([^']*)'") + + +# --------------------------------------------------------------------------- # +# Public entry point +# --------------------------------------------------------------------------- # + + +def activity_data_assets( + activity: AdfActivity, + definitions: AdfDefinitions, + context: TranslationContext | None = None, +) -> tuple[list[DataAsset], list[DataAsset]]: + """Resolve the assets an ADF activity reads and writes. + + Captures **every** input and output, not just index 0: a Copy reads its + ``source`` dataset(s) and writes its ``sink`` dataset(s), a Lookup reads its + ``dataset``, and any activity-level ``inputs`` / ``outputs`` slots are all + included. A dataset named in both an activity-level slot and ``typeProperties`` + is counted once per side so an activity does not emit two identical assets. + + Args: + activity: The ADF activity to resolve. + definitions: All loaded ADF definitions (datasets + linked services), + needed to resolve a reference to its physical identity. + context: Optional translation context for expression resolution; a default + (empty) context is used when none is given, mirroring the deterministic + discover-time resolution the closed #36 attempt used. + + Returns: + ``(data_reads, data_writes)`` as lists of :class:`DataAsset`. + """ + resolution_context = context if context is not None else TranslationContext() + reads = [ + _dataset_ref_to_asset(reference, definitions, resolution_context) + for reference in _activity_dataset_refs(activity, produced=False) + ] + writes = [ + _dataset_ref_to_asset(reference, definitions, resolution_context) + for reference in _activity_dataset_refs(activity, produced=True) + ] + return reads, writes + + +def _dataset_ref_to_asset( + dataset_ref: AdfDatasetReference, + definitions: AdfDefinitions, + context: TranslationContext, +) -> DataAsset: + """Turn one dataset reference into a two-tier :class:`DataAsset`. + + Signature is derived only from the resolved *physical* location -- the identity + when it resolves, else the structural path signature. It is **never** the bare + dataset reference name (#36's hard rule): two unrelated opaque references that + merely share a name must not join, so when neither a physical identity nor a + path-anchored signature is available the signature is left empty. An empty + signature is falsy, so :func:`~flowx.lineage._match_assets` cannot use it as a + join key -- the asset is still captured as a read / write for reporting, it just + cannot manufacture a signature-tier edge. + """ + identity = resolve_dataset_identity(dataset_ref, definitions, context) + if identity is not None: + # Mirror the identity into the signature so the weak tier never joins a + # resolved asset to an unresolved one that merely shares a physical value. + return DataAsset(signature=identity, identity=identity, asset_type=_asset_type(dataset_ref, definitions)) + path_signature = _path_signature(dataset_ref.parameters) + return DataAsset( + signature=path_signature if path_signature is not None else "", + identity=None, + asset_type=_asset_type(dataset_ref, definitions), + ) + + +# --------------------------------------------------------------------------- # +# Dataset reference gathering (all inputs / outputs) +# --------------------------------------------------------------------------- # + + +def _typeprops_dataset_ref(candidate: object) -> AdfDatasetReference | None: + """Build a dataset reference from a ``typeProperties`` source/sink/dataset slot. + + Carries the slot's ``parameters`` (Lookup / Delete / GetMetadata put the + dataset call-site params here) so the path signature can be computed. + """ + if isinstance(candidate, dict): + name = candidate.get("referenceName") + if isinstance(name, str) and name: + parameters = candidate.get("parameters") + return AdfDatasetReference( + reference_name=name, + parameters=parameters if isinstance(parameters, dict) else None, + ) + return None + + +def _activity_dataset_refs(activity: AdfActivity, *, produced: bool) -> Iterator[AdfDatasetReference]: + """Yield the dataset references an activity writes (produced) or reads (not). + + Gathers activity-level ``inputs`` / ``outputs`` **and** the ``typeProperties`` + ``source`` / ``sink`` / ``dataset`` slots. De-duplication is by + ``(reference_name, parameter binding)``, not by name alone: a dataset named in + both an activity slot and ``typeProperties`` with the *same* call-site params is + the same physical asset and is yielded once, but two uses of the *same* + parameterised dataset with *different* params (``ds(tbl=orders)`` vs + ``ds(tbl=customers)``) resolve to distinct physical assets and are both kept. + """ + type_properties = activity.type_properties or {} + candidates: list[AdfDatasetReference] = [] + if produced: + candidates.extend(activity.outputs or []) + sink_reference = _typeprops_dataset_ref(type_properties.get("sink")) + if sink_reference is not None: + candidates.append(sink_reference) + else: + candidates.extend(activity.inputs or []) + for key in ("source", "dataset"): + read_reference = _typeprops_dataset_ref(type_properties.get(key)) + if read_reference is not None: + candidates.append(read_reference) + + seen: set[tuple[str, str]] = set() + for reference in candidates: + dedupe_key = (reference.reference_name, _parameter_binding_key(reference)) + if dedupe_key in seen: + continue + seen.add(dedupe_key) + yield reference + + +def _parameter_binding_key(reference: AdfDatasetReference) -> str: + """Stable key for a reference's call-site parameter binding. + + Two references with the same name collapse only when their parameters match, so + distinct bindings that resolve to distinct physical assets survive. Sorted keys + make the string order-independent; ``default=str`` keeps it total for any value + an ADF export can carry. + """ + return json.dumps(reference.parameters or {}, sort_keys=True, default=str) + + +# --------------------------------------------------------------------------- # +# Physical identity resolution (tier 1) +# --------------------------------------------------------------------------- # + + +def resolve_dataset_identity( + dataset_ref: AdfDatasetReference, + definitions: AdfDefinitions, + context: TranslationContext | None = None, +) -> str | None: + """Deterministic physical identity for a dataset reference. + + Returns ``"schema.table"`` when a table is resolvable, else a storage path, + else ``None`` (never a guess). Used to join producers to consumers on the same + physical asset even when their ADF dataset names differ. + + Parameterised values (ADF expressions or DAB-ref placeholders) are treated as + unresolvable and return ``None`` -- they must never be used as identity keys + because two unrelated pipelines sharing a parameter name would collide on the + same placeholder string. + """ + resolution_context = context if context is not None else TranslationContext() + properties = _dataset_props(dataset_ref, definitions) + if properties is None: + return None + schema, table = _resolve_table_reference(dataset_ref, properties, resolution_context) + if table: + identity = f"{schema}.{table}" if schema else table + return identity if _is_physical(identity) else None + path = _resolve_dataset_path(properties, definitions) + return path if (path and _is_physical(path)) else None + + +def _dataset_props(dataset_ref: AdfDatasetReference, definitions: AdfDefinitions) -> dict[str, Any] | None: + """Return the ``properties`` dict for a dataset reference, or ``None``.""" + dataset = definitions.get_dataset(dataset_ref.reference_name) + if not dataset: + return None + return dict(dataset.properties or {}) + + +def _resolve_param_value( + raw: Any, + dataset_params: dict[str, Any], + context: TranslationContext, +) -> str: + """Resolve a single ADF location / table field to a string.""" + if raw is None: + return "" + if isinstance(raw, dict) and raw.get("type") == "Expression": + raw = raw.get("value", "") + if isinstance(raw, (list, dict)): + return "" + if not isinstance(raw, str): + return str(raw) + text = raw + + match = _DATASET_PARAM_RE.match(text.strip()) + if match: + parameter_name = match.group(1) + return _resolve_param_value(dataset_params.get(parameter_name, ""), dataset_params, context) + + if "@{" in text: + return resolve_interpolated_string(text, context) + + if text.startswith("@"): + result = resolve_expression(text, context) + if result is not None and result.kind in ("literal", "dab_ref"): + return result.value + return text + + return text + + +def _effective_dataset_params(dataset_ref: AdfDatasetReference, dataset_props: dict[str, Any]) -> dict[str, Any]: + """Effective parameter map: declared dataset defaults first, call-site overrides win.""" + declared = dataset_props.get("parameters") or {} + effective: dict[str, Any] = {} + for name, spec in declared.items(): + if isinstance(spec, dict) and "defaultValue" in spec: + effective[name] = spec["defaultValue"] + if dataset_ref.parameters: + effective.update(dict(dataset_ref.parameters)) + return effective + + +def _resolve_table_reference( + dataset_ref: AdfDatasetReference, + dataset_props: dict[str, Any] | None, + context: TranslationContext, +) -> tuple[str | None, str | None]: + """Resolve ``(schema, table)`` from a dataset reference. + + Handles both the nested ``typeProperties`` shape and the + ``schemaTypePropertiesSchema`` flattened form ``az datafactory dataset show`` + emits. ADF parameter expressions resolve against the reference's effective + parameter map. + """ + if not dataset_props: + return None, None + type_props = dataset_props.get("typeProperties") if isinstance(dataset_props.get("typeProperties"), dict) else None + effective_params = _effective_dataset_params(dataset_ref, dataset_props) + schema_raw = _pick_dataset_field( + type_props, + dataset_props, + ("schema", "database"), + ("schemaTypePropertiesSchema", "database"), + ) + table_raw = _pick_dataset_field( + type_props, + dataset_props, + ("table", "tableName"), + ("table", "tableName"), + ) + schema = _resolve_param_value(schema_raw, effective_params, context) if schema_raw is not None else None + table = _resolve_param_value(table_raw, effective_params, context) if table_raw is not None else None + return (schema or None), (table or None) + + +def _pick_dataset_field( + type_props: dict[str, Any] | None, + dataset_props: dict[str, Any], + nested_keys: tuple[str, ...], + flat_keys: tuple[str, ...], +) -> Any: + """First populated dataset field across the nested and az-flattened shapes. + + Empty strings, empty lists, and ``None`` are skipped so a column-schema + artifact like ``schema: []`` does not shadow the real database schema stored + under a flattened key. + """ + candidates: list[Any] = [] + if type_props is not None: + candidates.extend(type_props.get(key) for key in nested_keys) + candidates.extend(dataset_props.get(key) for key in flat_keys) + for value in candidates: + if value is None: + continue + if isinstance(value, (list, dict)) and not value: + continue + if isinstance(value, str) and not value.strip(): + continue + return value + return None + + +def _resolve_dataset_path(dataset_props: dict[str, Any], definitions: AdfDefinitions) -> str | None: + """Resolve a dataset's storage path from its location + backing linked service.""" + type_props = dataset_props.get("typeProperties") or dataset_props + location = type_props.get("location") or {} + if not isinstance(location, dict): + return None + + file_system = location.get("fileSystem") or location.get("container") or "" + folder_path = location.get("folderPath") or "" + if isinstance(file_system, dict) or isinstance(folder_path, dict): + return None # parameterised location; not a deterministic identity + + linked_service_reference = dataset_props.get("linkedServiceName") or {} + if isinstance(linked_service_reference, dict): + linked_service_name = linked_service_reference.get("referenceName", "") + else: + linked_service_name = str(linked_service_reference) + linked_service = definitions.get_linked_service(linked_service_name) if linked_service_name else None + account = _resolve_storage_account(linked_service) + if not account: + return None + + return f"abfss://{file_system}@{account}.dfs.core.windows.net/{folder_path}".rstrip("/") + + +def _resolve_storage_account(linked_service: Any) -> str | None: + """Pull a storage account name out of a linked service, if present.""" + if linked_service is None: + return None + type_props = linked_service.properties.get("typeProperties") or linked_service.properties + + url = type_props.get("url") or "" + if isinstance(url, str) and url: + host = url.replace("https://", "").split("/", 1)[0] + host_no_port = host.split(":", 1)[0] + if "." in host_no_port: + return host_no_port.split(".", 1)[0] + + sas_uri = type_props.get("sasUri") or "" + if isinstance(sas_uri, str) and sas_uri: + host = sas_uri.split("?", 1)[0].replace("https://", "").split("/", 1)[0] + if "." in host: + return host.split(".", 1)[0] + + # Plaintext connection string (rare in az exports -- usually masked). + connection_string = type_props.get("connectionString") + if isinstance(connection_string, str): + match = _ACCOUNT_NAME_RE.search(connection_string) + if match: + return match.group(1) + if isinstance(connection_string, dict): + value = connection_string.get("value", "") + match = _ACCOUNT_NAME_RE.search(value) + if match: + return match.group(1) + + # AWS -- bucket name lives on the dataset, account is implicit; nothing useful + # to return at the linked-service level for S3 / GCS. + return None + + +def _is_physical(value: str) -> bool: + """Return ``True`` only when *value* is a literal (physical) identifier. + + A value is NOT physical when it still contains an unresolved marker -- a + DAB-ref placeholder (``{{`` ... ``}}``), a leftover ADF interpolation + fragment (``@{``), or a bare ADF expression (starts with ``@``). + """ + stripped = value.lstrip() + return not ("{{" in value or "@{" in value or stripped.startswith("@")) + + +# --------------------------------------------------------------------------- # +# Structural path signature (tier 2, #36's "expression" tier) +# --------------------------------------------------------------------------- # + + +def _path_signature(parameters: dict[str, Any] | None) -> str | None: + """Structural signature of a reference's parameterised folderPath / fileName. + + Requires a resolvable **folderPath** literal anchor. A file name alone is too + weak a discriminator: many unrelated activities write ``.csv`` / ``.json`` files + to opaque parameterised folders, so a signature built only from a file extension + would join them all -- re-creating the explosion the identity-only join avoids. + Anchoring on the literal folder segment keeps the match specific to a real, + named location. + + ``None`` when the folder path has no literal segment to anchor on. + """ + if not parameters: + return None + folder_signature = _normalize_path_expression(parameters.get("folderPath")) + if folder_signature is None: + return None + file_signature = _normalize_path_expression(parameters.get("fileName")) + return f"FP[{folder_signature}]/FN[{file_signature}]" + + +def _normalize_path_expression(expression: Any) -> str | None: + """Reduce a (possibly parameterised) ADF path expression to a structural signature. + + Keeps the literal path segments and collapses every runtime reference to a + single ``

`` slot, so the result captures the path *shape* (literal skeleton + plus slot count) without guessing the runtime value. Returns ``None`` when + there is no literal segment to anchor on (a signature of only slots is too weak + a join key -- never guess). + """ + if isinstance(expression, dict): + expression = expression.get("value", "") + if not isinstance(expression, str) or not expression.strip(): + return None + text = expression.strip() + if "@" not in text: + # A bare literal value (no ADF expression): the whole string is the literal + # path / filename, with no runtime slots. + literal = re.sub(r"/+", "/", text).strip("/") + return f"{literal}|slots=0" if literal else None + marked = _PARAM_REF_RE.sub("

", text) + literal = re.sub(r"/+", "/", "".join(_QUOTED_LITERAL_RE.findall(marked))).strip("/") + if not literal: + return None + return f"{literal}|slots={marked.count('

')}" + + +# --------------------------------------------------------------------------- # +# Asset-type classification (best-effort neutral kind) +# --------------------------------------------------------------------------- # + +_TABLE_HINTS = ("table", "sql", "database") +_FILE_HINTS = ("delimited", "parquet", "orc", "avro", "json", "binary", "excel", "xml", "blob", "adls", "file") + + +def _asset_type(dataset_ref: AdfDatasetReference, definitions: AdfDefinitions) -> str | None: + """Best-effort neutral asset kind (``"table"`` / ``"file"``), or ``None``. + + Prefers the shape of the dataset's typeProperties (a ``location`` block means a + file, a ``table`` / ``schema`` means a table) and falls back to keyword hints in + the dataset's ADF type string. Returns ``None`` when nothing is conclusive + rather than guessing. + """ + dataset = definitions.get_dataset(dataset_ref.reference_name) + if dataset is None: + return None + type_props = dataset.properties.get("typeProperties") + if isinstance(type_props, dict): + if isinstance(type_props.get("location"), dict): + return "file" + if type_props.get("table") or type_props.get("tableName") or type_props.get("schema"): + return "table" + dataset_type = (dataset.type or "").lower() + if any(hint in dataset_type for hint in _TABLE_HINTS): + return "table" + if any(hint in dataset_type for hint in _FILE_HINTS): + return "file" + return None diff --git a/src/flowx/sources/adf/discovery_mapping.py b/src/flowx/sources/adf/discovery_mapping.py new file mode 100644 index 00000000..0555a312 --- /dev/null +++ b/src/flowx/sources/adf/discovery_mapping.py @@ -0,0 +1,394 @@ +"""Map the ADF AST onto the shared, source-faithful discovery AST. + +This is the ADF half of the discovery contract (issue #62): it turns the typed +ADF AST (:mod:`flowx.models.adf_ast`) into the shared +:class:`~flowx.models.discovery.SourceGraph` model that both ADF and Airflow +align to. The mapping is deliberately **1:1 and lossless**: + +* every ADF activity becomes exactly one discovery node -- no motif collapse, + no merging; +* the activity's own type string is kept verbatim as ``native_type``; +* the verbatim source dict is preserved on every node (``raw``) and on the graph; +* dependency edges keep **all** of their outcome conditions (a ``dependsOn`` with + ``["Succeeded", "Skipped"]`` stays a two-condition edge); +* anything without a shared typed home -- the ADF folder, retry policy detail, + the target-side translation strategy -- rides in the ``properties`` / + ``extensions`` seams rather than being dropped or forced into a field. + +Data lineage (#62b) is populated here now: every node's ``data_reads`` / +``data_writes`` are resolved from the ADF dataset references via +:func:`~flowx.sources.adf.dataset_lineage.activity_data_assets` (the two-tier +identity / signature model), and every ``ExecutePipeline`` node records the child +pipeline it invokes under the neutral +:data:`~flowx.discovery_lineage.INVOKES_WORKFLOW_PROPERTY` marker. Because +:func:`_activity_to_node` recurses into every control-flow branch, this population +is Switch-aware for free -- a Copy or ExecutePipeline nested inside a Switch case +(or default), a ForEach / Until body, or either If branch carries its assets and +marker like any top-level node. The graph's :attr:`SourceGraph.lineage` block is +then derived by the source-neutral :func:`~flowx.discovery_lineage.build_graph_lineage`. + +The control-flow container shape follows :class:`~flowx.models.discovery.ContainerNode`: +an ``IfCondition`` becomes ``{"true": [...], "false": [...]}``, a ``ForEach`` / +``Until`` becomes ``{"body": [...]}``, and a ``Switch`` becomes +``{"": [...], "default": [...]}``. Branch insertion order matches the +order ADF declares the branches so a downstream flatten reproduces source order. +""" + +from __future__ import annotations + +from typing import Any + +from flowx.discovery_inventory import STRATEGY_PROPERTY +from flowx.discovery_lineage import INVOKES_WAIT_PROPERTY, INVOKES_WORKFLOW_PROPERTY, build_graph_lineage +from flowx.models.adf_ast import AdfActivity, AdfDefinitions, AdfParameter, AdfPipeline, AdfTrigger, AdfVariable +from flowx.models.discovery import ( + CONCEPT_BRANCH, + CONCEPT_COPY_DATA, + CONCEPT_GAP, + CONCEPT_LOOP, + CONCEPT_NOTEBOOK, + CONCEPT_QUERY, + CONCEPT_RUN_WORKFLOW, + CONCEPT_SCRIPT, + CONCEPT_SET_VARIABLE, + CONCEPT_SWITCH, + CONCEPT_WAIT, + SOURCE_ADF, + ContainerNode, + GapNode, + ParameterSpec, + PolicySpec, + ScheduleSpec, + SourceDependency, + SourceGraph, + SourceNode, +) +from flowx.sources.adf.dataset_lineage import activity_data_assets +from flowx.sources.adf.loader import classify_activity + +# Neutral trigger category for each ADF trigger type. Anything unrecognised maps +# to ``""`` (unknown) with the ADF type preserved verbatim in the schedule +# extensions, so no trigger information is lost even for a type not listed here. +_TRIGGER_KIND: dict[str, str] = { + "ScheduleTrigger": "schedule", + "TumblingWindowTrigger": "interval", + "BlobEventsTrigger": "file_arrival", + "CustomEventsTrigger": "event", +} + +# Neutral concept for each ADF activity type. The discovery concept vocabulary is +# intentionally non-exhaustive: ADF types with no shared concept (WebActivity, +# Delete, Filter, and the agentic-only types) map to CONCEPT_GAP, which here means +# "no shared concept applies" -- independent of the translation strategy, which is +# recorded separately under properties[STRATEGY_PROPERTY]. +_CONCEPT_BY_TYPE: dict[str, str] = { + "Copy": CONCEPT_COPY_DATA, + "DatabricksNotebook": CONCEPT_NOTEBOOK, + "DatabricksSparkJar": CONCEPT_SCRIPT, + "DatabricksSparkPython": CONCEPT_SCRIPT, + "DatabricksJob": CONCEPT_RUN_WORKFLOW, + "ExecutePipeline": CONCEPT_RUN_WORKFLOW, + "ForEach": CONCEPT_LOOP, + "Until": CONCEPT_LOOP, + "IfCondition": CONCEPT_BRANCH, + "Switch": CONCEPT_SWITCH, + "SetVariable": CONCEPT_SET_VARIABLE, + "AppendVariable": CONCEPT_SET_VARIABLE, + "Wait": CONCEPT_WAIT, + "Lookup": CONCEPT_QUERY, +} + + +def adf_definitions_to_source_graphs(definitions: AdfDefinitions) -> list[SourceGraph]: + """Map every pipeline in *definitions* to a shared :class:`SourceGraph`. + + Pipeline order is preserved so downstream consumers see the same ordering the + loader produced. ADF triggers are mapped into :class:`ScheduleSpec` and attached + to each pipeline they reference (see :func:`_attach_schedules`), so schedule / + trigger information is preserved rather than dropped. Each graph's + :attr:`SourceGraph.lineage` block is derived (control + data edges) once its + nodes' reads / writes and invocation markers are populated -- data edges are + joined within a single graph, matching the per-workflow scope of the shared + :class:`~flowx.models.ir.Lineage` model. + + ``definitions`` is threaded down to the activity mapper so a node's data assets + resolve against the factory's datasets and linked services. + """ + graphs = [adf_pipeline_to_source_graph(pipeline, definitions) for pipeline in definitions.pipelines] + _attach_schedules(graphs, definitions.triggers) + for graph in graphs: + graph.lineage = build_graph_lineage(graph) + return graphs + + +def _attach_schedules(graphs: list[SourceGraph], triggers: list[AdfTrigger]) -> None: + """Populate each graph's ``schedule`` from the triggers that reference it. + + A trigger can drive several pipelines and a pipeline can be driven by several + triggers. The first trigger to reference a pipeline becomes its typed + :attr:`SourceGraph.schedule`; any further triggers for the same pipeline are + preserved verbatim under ``schedule.extensions["additional_triggers"]`` so + nothing is lost. Pipeline references are matched case-insensitively, mirroring + ADF's case-insensitive identifier semantics. + + A **fresh** :class:`ScheduleSpec` is built for each pipeline assignment rather + than sharing one instance across every pipeline a trigger references: a shared + instance would let a later trigger's mutation (appending to + ``additional_triggers``) leak onto every other pipeline that trigger touched. + """ + graphs_by_name = {graph.name: graph for graph in graphs} + graphs_by_lower = {graph.name.lower(): graph for graph in graphs} + + for trigger in triggers: + for pipeline_name in _trigger_pipeline_names(trigger): + graph = graphs_by_name.get(pipeline_name) or graphs_by_lower.get(pipeline_name.lower()) + if graph is None: + continue + if graph.schedule is None: + # Per-pipeline instance: no shared mutable state across pipelines. + graph.schedule = _trigger_to_schedule(trigger) + else: + additional = graph.schedule.extensions.setdefault("additional_triggers", []) + additional.append( + { + "trigger_name": trigger.name, + "trigger_type": trigger.type, + "properties": trigger.properties, + } + ) + + +def _trigger_to_schedule(trigger: AdfTrigger) -> ScheduleSpec: + """Map a single ADF trigger to a source-faithful :class:`ScheduleSpec`. + + The recurrence payload is kept as-given: schedule triggers nest it under + ``typeProperties.recurrence`` while tumbling-window / event triggers put their + detail directly in ``typeProperties``, so whichever is present becomes the + verbatim ``expression``. The full trigger ``properties`` block also rides in + ``extensions`` so the mapping is lossless even for trigger detail with no typed + home yet (pipeline parameters, runtime state, annotations). + """ + type_properties = trigger.properties.get("typeProperties") or {} + recurrence = type_properties.get("recurrence") if isinstance(type_properties, dict) else None + if isinstance(recurrence, dict): + expression: Any = recurrence + timezone = recurrence.get("timeZone") + else: + expression = type_properties or None + timezone = type_properties.get("timeZone") if isinstance(type_properties, dict) else None + + return ScheduleSpec( + kind=_TRIGGER_KIND.get(trigger.type, ""), + expression=expression, + timezone=timezone, + extensions={ + "trigger_name": trigger.name, + "trigger_type": trigger.type, + "properties": trigger.properties, + }, + ) + + +def _trigger_pipeline_names(trigger: AdfTrigger) -> list[str]: + """Collect the names of the pipelines a trigger references, in order.""" + names: list[str] = [] + for reference in trigger.pipelines or []: + pipeline_reference = reference.get("pipelineReference") if isinstance(reference, dict) else None + name = pipeline_reference.get("referenceName") if isinstance(pipeline_reference, dict) else None + if name: + names.append(name) + return names + + +def adf_pipeline_to_source_graph(pipeline: AdfPipeline, definitions: AdfDefinitions | None = None) -> SourceGraph: + """Map a single ADF pipeline to a source-faithful :class:`SourceGraph`. + + ``definitions`` supplies the datasets and linked services the activity mapper + needs to resolve each node's data assets to a physical identity. It is optional + so a caller mapping a pipeline in isolation still works; without it, dataset + references simply stay unresolved (identity ``None``) and fall back to their + neutral signature. It does **not** attach the graph-level lineage block -- + :func:`adf_definitions_to_source_graphs` owns that, once every graph is built. + """ + resolved_definitions = definitions if definitions is not None else AdfDefinitions(pipelines=[]) + properties: dict[str, str] = {} + if pipeline.folder: + properties["folder"] = pipeline.folder + + return SourceGraph( + name=pipeline.name, + source=SOURCE_ADF, + parameters=_parameters_to_specs(pipeline.parameters), + variables=_variables_to_specs(pipeline.variables), + tags=list(pipeline.annotations) if pipeline.annotations else [], + tasks=[_activity_to_node(activity, resolved_definitions) for activity in pipeline.activities], + properties=properties, + raw=pipeline.raw, + ) + + +def _activity_to_node(activity: AdfActivity, definitions: AdfDefinitions) -> SourceNode: + """Map one ADF activity to a discovery node (1:1, source-faithful). + + Fields are passed explicitly to each node class (rather than unpacking a + shared dict) so the mapping stays type-checked end to end. Data reads / writes + are resolved from the activity's dataset references, and an ``ExecutePipeline`` + records the child pipeline it invokes under the neutral control-edge marker; + both apply to nested nodes too because the branch children below route back + through this function. + """ + strategy = classify_activity(activity.type) + concept = _CONCEPT_BY_TYPE.get(activity.type, CONCEPT_GAP) + dependencies = [ + SourceDependency(upstream=dependency.activity, conditions=list(dependency.dependency_conditions)) + for dependency in (activity.depends_on or []) + ] + policy = _policy_to_spec(activity) + properties: dict[str, Any] = {STRATEGY_PROPERTY: strategy.value} + _record_invocation(activity, properties) + data_reads, data_writes = activity_data_assets(activity, definitions) + + branches = _control_flow_branches(activity, definitions) + if branches is not None: + return ContainerNode( + source_id=activity.name, + task_key=activity.name, + concept=concept, + source=SOURCE_ADF, + name=activity.name, + native_type=activity.type, + dependencies=dependencies, + policy=policy, + data_reads=data_reads, + data_writes=data_writes, + properties=properties, + raw=activity.raw, + branches=branches, + ) + if strategy.value == "unsupported": + return GapNode( + source_id=activity.name, + task_key=activity.name, + source=SOURCE_ADF, + name=activity.name, + native_type=activity.type, + dependencies=dependencies, + policy=policy, + data_reads=data_reads, + data_writes=data_writes, + properties=properties, + raw=activity.raw, + reason=f"unsupported ADF activity type {activity.type!r}", + ) + return SourceNode( + source_id=activity.name, + task_key=activity.name, + concept=concept, + source=SOURCE_ADF, + name=activity.name, + native_type=activity.type, + dependencies=dependencies, + policy=policy, + data_reads=data_reads, + data_writes=data_writes, + properties=properties, + raw=activity.raw, + ) + + +def _record_invocation(activity: AdfActivity, properties: dict[str, Any]) -> None: + """Stash the child pipeline an ``ExecutePipeline`` invokes under neutral keys. + + The callee reference name and ``waitOnCompletion`` flag are read from the same + ADF ``typeProperties`` the convert-time translator reads, then recorded under + the source-neutral :data:`INVOKES_WORKFLOW_PROPERTY` / :data:`INVOKES_WAIT_PROPERTY` + keys so the source-agnostic control-edge derivation can find them without + knowing anything about ADF. An empty / missing callee still records the marker + (with an empty target) so the edge is reported as unresolved rather than dropped. + """ + if activity.type != "ExecutePipeline": + return + type_properties = activity.type_properties or {} + reference = type_properties.get("pipeline", {}) + if isinstance(reference, dict): + callee = reference.get("referenceName", "") or "" + else: + callee = str(reference) + properties[INVOKES_WORKFLOW_PROPERTY] = callee + properties[INVOKES_WAIT_PROPERTY] = bool(type_properties.get("waitOnCompletion", True)) + + +def _control_flow_branches(activity: AdfActivity, definitions: AdfDefinitions) -> dict[str, list[SourceNode]] | None: + """Return the labelled branches of a control-flow activity, or ``None`` if it is a leaf. + + Keyed on the activity **type**, not on whether children happen to be present: + a control-flow activity always maps to a :class:`ContainerNode`, and every + branch it declares is always represented -- an empty branch is present-but-empty, + never omitted. So an empty ``IfCondition`` still yields ``{"true": [], "false": []}`` + and a one-sided ``If`` keeps its empty ``false`` branch, rather than collapsing to + a plain node and losing the control-flow structure. Returning ``None`` (not an + empty dict) is what tells the caller the activity is a leaf. + + Branch order matches ADF's own declaration order (true before false, cases + before default) so a downstream flatten walks children in source order. + """ + if activity.type == "IfCondition": + return { + "true": [_activity_to_node(child, definitions) for child in (activity.if_true_activities or [])], + "false": [_activity_to_node(child, definitions) for child in (activity.if_false_activities or [])], + } + if activity.type in ("ForEach", "Until"): + return {"body": [_activity_to_node(child, definitions) for child in (activity.activities or [])]} + if activity.type == "Switch": + branches: dict[str, list[SourceNode]] = {} + for case_value, case_activities in (activity.switch_cases or {}).items(): + branches[case_value] = [_activity_to_node(child, definitions) for child in case_activities] + branches["default"] = [ + _activity_to_node(child, definitions) for child in (activity.switch_default_activities or []) + ] + return branches + return None + + +def _policy_to_spec(activity: AdfActivity) -> PolicySpec | None: + """Map an ADF retry/timeout policy to a :class:`PolicySpec`, if present. + + Retry count and interval are already numeric in ADF and map to typed fields. + The ADF timeout is an ISO-8601 duration *string* -- normalising it to seconds + is a target concern, so it rides verbatim in ``extensions`` alongside the + ``secureInput`` / ``secureOutput`` flags rather than being guessed at here. + """ + policy = activity.policy + if policy is None: + return None + extensions: dict[str, object] = {} + if policy.timeout is not None: + extensions["timeout"] = policy.timeout + if policy.secure_input: + extensions["secure_input"] = True + if policy.secure_output: + extensions["secure_output"] = True + return PolicySpec( + max_retries=policy.retry, + retry_interval_seconds=policy.retry_interval_in_seconds, + extensions=extensions, + ) + + +def _parameters_to_specs(parameters: dict[str, AdfParameter] | None) -> dict[str, ParameterSpec]: + """Map ADF pipeline parameters to shared parameter specs.""" + if not parameters: + return {} + return { + name: ParameterSpec(type=parameter.type, default=parameter.default_value) + for name, parameter in parameters.items() + } + + +def _variables_to_specs(variables: dict[str, AdfVariable] | None) -> dict[str, ParameterSpec]: + """Map ADF pipeline variables to shared parameter specs (same shape).""" + if not variables: + return {} + return { + name: ParameterSpec(type=variable.type, default=variable.default_value) for name, variable in variables.items() + } diff --git a/src/flowx/sources/adf/loader.py b/src/flowx/sources/adf/loader.py index 5fcc5d67..27ce7c1b 100644 --- a/src/flowx/sources/adf/loader.py +++ b/src/flowx/sources/adf/loader.py @@ -8,7 +8,6 @@ import logging import re import shutil -from datetime import datetime, timezone from pathlib import Path from typing import Any @@ -29,6 +28,7 @@ InventoryItem, TranslationStrategy, ) +from flowx.models.motifs import DetectedMotif logger = logging.getLogger(__name__) @@ -789,50 +789,6 @@ def _classify_activities( _classify_activities(pipeline_name, switch_children, items) -# --------------------------------------------------------------------------- -# Serialisation helpers -# --------------------------------------------------------------------------- - - -def _inventory_to_dict(inventory: Inventory, source_dir: str) -> dict[str, Any]: - """Serialise an :class:`Inventory` to a JSON-friendly dictionary. - - Args: - inventory: The inventory to serialise. - source_dir: Original source directory path (for provenance). - - Returns: - Dictionary suitable for ``json.dumps``. - """ - pipeline_map: dict[str, list[dict[str, Any]]] = {} - for item in inventory.items: - entry: dict[str, Any] = { - "name": item.activity_name, - "type": item.activity_type, - "strategy": item.strategy.value, - } - if item.depends_on: - entry["depends_on"] = item.depends_on - pipeline_map.setdefault(item.pipeline_name, []).append(entry) - - total = inventory.deterministic_count + inventory.agentic_count + inventory.unsupported_count - coverage_pct = round((inventory.deterministic_count + inventory.agentic_count) / total * 100, 1) if total else 0.0 - - return { - "source_dir": source_dir, - "generated_at": datetime.now(timezone.utc).isoformat(), - "pipelines": [{"name": pname, "activities": acts} for pname, acts in pipeline_map.items()], - "summary": { - "pipeline_count": inventory.pipeline_count, - "activity_count": total, - "deterministic_count": inventory.deterministic_count, - "agentic_count": inventory.agentic_count, - "unsupported_count": inventory.unsupported_count, - "coverage_pct": coverage_pct, - }, - } - - # --------------------------------------------------------------------------- # Profile complexity report (CSV) # --------------------------------------------------------------------------- @@ -924,7 +880,33 @@ def _tshirt_size(score: int) -> str: return "XL" -def build_profile_rows(definitions: AdfDefinitions) -> list[dict[str, Any]]: +def detect_motifs_by_pipeline(definitions: AdfDefinitions) -> dict[str, list[DetectedMotif]]: + """Detect the motifs in every pipeline, keyed by pipeline name. + + Runs the profiler's motif detector (:func:`flowx.motifs.detector.detect_motifs`) + once per pipeline, so the discover phase can both surface the full detected + motifs additively in ``inventory.json`` and count them in the profile report + from a single detection pass. Detection is best-effort: a pipeline whose + detection raises is logged and recorded with an empty list, never allowed to + hard-fail discover. This only *detects* motifs -- it never collapses their + member activities, which stays a convert-phase decision. + """ + from flowx.motifs.detector import detect_motifs + + results: dict[str, list[DetectedMotif]] = {} + for pipeline in definitions.pipelines: + try: + results[pipeline.name] = detect_motifs(pipeline, definitions) + except Exception as exc: # noqa: BLE001 - detection must never hard-fail discover + logger.warning("Motif detection failed for pipeline %r: %s", pipeline.name, exc) + results[pipeline.name] = [] + return results + + +def build_profile_rows( + definitions: AdfDefinitions, + motifs_by_pipeline: dict[str, list[DetectedMotif]] | None = None, +) -> list[dict[str, Any]]: """Builds one profile-report row per pipeline. Each row carries the source activity / dataset / linked-service counts, the @@ -933,20 +915,22 @@ def build_profile_rows(definitions: AdfDefinitions) -> list[dict[str, Any]]: Args: definitions: Parsed ADF definitions. + motifs_by_pipeline: Optional precomputed motif detections keyed by + pipeline name (see :func:`detect_motifs_by_pipeline`). Passing the + same map the inventory emitter uses keeps the report's pattern count + consistent with the surfaced motifs and avoids detecting twice; when + omitted, detection runs here. Returns: List of row dicts ordered by pipeline name. """ - from flowx.motifs.detector import detect_motifs + if motifs_by_pipeline is None: + motifs_by_pipeline = detect_motifs_by_pipeline(definitions) rows: list[dict[str, Any]] = [] for pipeline in sorted(definitions.pipelines, key=lambda p: p.name): activity_count, datasets, linked_services, category_counts = _pipeline_reference_counts(pipeline, definitions) - try: - n_patterns = len(detect_motifs(pipeline, definitions)) - except Exception as exc: # noqa: BLE001 - profiling must never hard-fail on motif detection - logger.warning("Motif detection failed for pipeline %r: %s", pipeline.name, exc) - n_patterns = 0 + n_patterns = len(motifs_by_pipeline.get(pipeline.name, [])) score = _complexity_score(category_counts, len(datasets), len(linked_services), n_patterns) rows.append( { @@ -1095,7 +1079,12 @@ def main(argv: list[str] | None = None) -> int: ) logger.info("Filtered to pipeline: %s", args.pipeline) - inventory = build_inventory(definitions) + # Inventory JSON is projected from the shared discovery AST via the + # source-agnostic emitter (so ADF and Airflow emit one shape); imported here + # to avoid a module-level cycle (discovery_mapping imports this loader). + from flowx.discovery_inventory import build_source_inventory + from flowx.models.discovery import SOURCE_ADF + from flowx.sources.adf.discovery_mapping import adf_definitions_to_source_graphs output_dir: Path = args.output_dir.resolve() clear_stale_outputs(output_dir) @@ -1103,11 +1092,23 @@ def main(argv: list[str] | None = None) -> int: metadata_dir.mkdir(parents=True, exist_ok=True) inventory_path = metadata_dir / "inventory.json" - inventory_dict = _inventory_to_dict(inventory, str(args.source_dir)) + source_graphs = adf_definitions_to_source_graphs(definitions) + # Detect motifs once and share the result: the inventory surfaces the full + # detections additively, the profile report counts them -- from one pass. + motifs_by_pipeline = detect_motifs_by_pipeline(definitions) + inventory_dict = build_source_inventory( + source_graphs, + source=SOURCE_ADF, + source_dir=str(args.source_dir), + # ADF has historically omitted zero-activity pipelines from the per-pipeline + # listing while still counting them in summary.pipeline_count; preserve that. + include_empty_pipelines=False, + motifs_by_pipeline=motifs_by_pipeline, + ) inventory_path.write_text(json.dumps(inventory_dict, indent=2), encoding="utf-8") logger.info("Wrote inventory to %s", inventory_path) - profile_rows = build_profile_rows(definitions) + profile_rows = build_profile_rows(definitions, motifs_by_pipeline) csv_path = metadata_dir / "profile_report.csv" write_profile_csv(profile_rows, csv_path) logger.info("Wrote profile report to %s", csv_path) diff --git a/tests/resources/golden/adf_fixture_coverage.json b/tests/resources/golden/adf_fixture_coverage.json new file mode 100644 index 00000000..c36912da --- /dev/null +++ b/tests/resources/golden/adf_fixture_coverage.json @@ -0,0 +1,1162 @@ +[ + { + "activities": 13, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 13, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 31, + "complexity_size": "XL", + "control_flow_activities": 4, + "coverage_pct": 100.0, + "databricks_native_activities": 4, + "datasets": 3, + "deterministic_activities": 13, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 1, + "migration_status": "included", + "other_activities": 5, + "pipeline": "pipeline_all_activity_types", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 8, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 8, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 1, + "complexity_score": 23, + "complexity_size": "L", + "control_flow_activities": 3, + "coverage_pct": 100.0, + "databricks_native_activities": 1, + "datasets": 2, + "deterministic_activities": 8, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 1, + "migration_status": "included", + "other_activities": 4, + "pipeline": "pipeline_all_dependency_conditions", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 5, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 5, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 1, + "complexity_score": 14, + "complexity_size": "M", + "control_flow_activities": 3, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 1, + "deterministic_activities": 5, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 2, + "pipeline": "pipeline_append_variable_loop", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 8, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 8, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 3, + "complexity_score": 26, + "complexity_size": "L", + "control_flow_activities": 1, + "coverage_pct": 100.0, + "databricks_native_activities": 3, + "datasets": 4, + "deterministic_activities": 8, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 2, + "migration_status": "included", + "other_activities": 4, + "pipeline": "pipeline_complex_etl", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 15, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 15, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 1, + "complexity_score": 39, + "complexity_size": "XL", + "control_flow_activities": 7, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 0, + "deterministic_activities": 15, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 8, + "pipeline": "pipeline_complex_orchestration", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 1, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 1, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 7, + "complexity_size": "M", + "control_flow_activities": 0, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 2, + "deterministic_activities": 1, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 2, + "migration_status": "included", + "other_activities": 1, + "pipeline": "pipeline_copy_csv_to_delta", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 1, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 1, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 6, + "complexity_size": "M", + "control_flow_activities": 0, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 2, + "deterministic_activities": 1, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 1, + "migration_status": "included", + "other_activities": 1, + "pipeline": "pipeline_copy_parquet_to_delta", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 1, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 1, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 6, + "complexity_size": "M", + "control_flow_activities": 0, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 2, + "deterministic_activities": 1, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 1, + "migration_status": "included", + "other_activities": 1, + "pipeline": "pipeline_copy_sql_to_delta", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 2, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 2, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 8, + "complexity_size": "M", + "control_flow_activities": 0, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 2, + "deterministic_activities": 2, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 2, + "pipeline": "pipeline_delete_recursive", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 3, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 3, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 9, + "complexity_size": "M", + "control_flow_activities": 0, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 0, + "deterministic_activities": 3, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 3, + "pipeline": "pipeline_execute_pipeline_nested", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 4, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 4, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 10, + "complexity_size": "M", + "control_flow_activities": 2, + "coverage_pct": 100.0, + "databricks_native_activities": 1, + "datasets": 1, + "deterministic_activities": 4, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 1, + "migration_status": "included", + "other_activities": 1, + "pipeline": "pipeline_filter_array", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 7, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 7, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 18, + "complexity_size": "L", + "control_flow_activities": 3, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 0, + "deterministic_activities": 7, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 4, + "pipeline": "pipeline_foreach_switch", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 2, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 2, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 7, + "complexity_size": "M", + "control_flow_activities": 1, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 2, + "deterministic_activities": 2, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 1, + "pipeline": "pipeline_foreach_with_copy", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 5, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 5, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 15, + "complexity_size": "M", + "control_flow_activities": 1, + "coverage_pct": 100.0, + "databricks_native_activities": 1, + "datasets": 2, + "deterministic_activities": 5, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 1, + "migration_status": "included", + "other_activities": 3, + "pipeline": "pipeline_if_condition_branching", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 3, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 3, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 1, + "complexity_score": 12, + "complexity_size": "M", + "control_flow_activities": 1, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 3, + "deterministic_activities": 3, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 2, + "pipeline": "pipeline_lookup_and_foreach", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 6, + "agentic_activities": 4, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 6, + "code_attached_coverage_pct": 83.3, + "collapsible_patterns": 0, + "complexity_score": 19, + "complexity_size": "L", + "control_flow_activities": 0, + "coverage_pct": 83.3, + "databricks_native_activities": 0, + "datasets": 1, + "deterministic_activities": 1, + "deterministic_coverage_pct": 16.7, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 6, + "pipeline": "pipeline_mixed_agentic", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 4, + "unresolved_agentic_count": 0, + "unsupported_activities": 1 + }, + { + "activities": 6, + "agentic_activities": 1, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 6, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 1, + "complexity_score": 20, + "complexity_size": "L", + "control_flow_activities": 1, + "coverage_pct": 100.0, + "databricks_native_activities": 1, + "datasets": 3, + "deterministic_activities": 5, + "deterministic_coverage_pct": 83.3, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 1, + "migration_status": "included", + "other_activities": 4, + "pipeline": "pipeline_mixed_deterministic_agentic", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 1, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 1, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 1, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 2, + "complexity_size": "S", + "control_flow_activities": 0, + "coverage_pct": 100.0, + "databricks_native_activities": 1, + "datasets": 0, + "deterministic_activities": 1, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 1, + "migration_status": "included", + "other_activities": 0, + "pipeline": "pipeline_notebook_basic", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 1, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 1, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 2, + "complexity_size": "S", + "control_flow_activities": 0, + "coverage_pct": 100.0, + "databricks_native_activities": 1, + "datasets": 0, + "deterministic_activities": 1, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 1, + "migration_status": "included", + "other_activities": 0, + "pipeline": "pipeline_notebook_with_params", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 5, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 5, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 10, + "complexity_size": "M", + "control_flow_activities": 4, + "coverage_pct": 100.0, + "databricks_native_activities": 1, + "datasets": 0, + "deterministic_activities": 5, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 1, + "migration_status": "included", + "other_activities": 0, + "pipeline": "pipeline_set_variable_chain", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 1, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 1, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 2, + "complexity_size": "S", + "control_flow_activities": 0, + "coverage_pct": 100.0, + "databricks_native_activities": 1, + "datasets": 0, + "deterministic_activities": 1, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 1, + "migration_status": "included", + "other_activities": 0, + "pipeline": "pipeline_spark_jar_job", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 2, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 2, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 4, + "complexity_size": "S", + "control_flow_activities": 0, + "coverage_pct": 100.0, + "databricks_native_activities": 2, + "datasets": 0, + "deterministic_activities": 2, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 2, + "migration_status": "included", + "other_activities": 0, + "pipeline": "pipeline_spark_python_job", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 7, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 7, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 23, + "complexity_size": "L", + "control_flow_activities": 2, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 4, + "deterministic_activities": 7, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 5, + "pipeline": "pipeline_switch_multi_case", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 5, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 5, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 12, + "complexity_size": "M", + "control_flow_activities": 2, + "coverage_pct": 100.0, + "databricks_native_activities": 1, + "datasets": 0, + "deterministic_activities": 5, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 1, + "migration_status": "included", + "other_activities": 2, + "pipeline": "pipeline_wait_between_steps", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 3, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 3, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 9, + "complexity_size": "M", + "control_flow_activities": 0, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 0, + "deterministic_activities": 3, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 3, + "pipeline": "pipeline_web_activity_auth", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 3, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 3, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 6, + "complexity_size": "M", + "control_flow_activities": 3, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 0, + "deterministic_activities": 3, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 0, + "pipeline": "pl_test_appendvariable_coverage", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 2, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 2, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 10, + "complexity_size": "M", + "control_flow_activities": 0, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 4, + "deterministic_activities": 2, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 2, + "pipeline": "pl_test_copy_coverage", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 1, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 1, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 4, + "complexity_size": "S", + "control_flow_activities": 0, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 1, + "deterministic_activities": 1, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 1, + "pipeline": "pl_test_delete_coverage", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 2, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 2, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 6, + "complexity_size": "M", + "control_flow_activities": 0, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 0, + "deterministic_activities": 2, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 2, + "pipeline": "pl_test_executepipeline_coverage", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 4, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 4, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 10, + "complexity_size": "M", + "control_flow_activities": 3, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 1, + "deterministic_activities": 4, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 1, + "pipeline": "pl_test_filter_coverage", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 2, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 2, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 7, + "complexity_size": "M", + "control_flow_activities": 1, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 2, + "deterministic_activities": 2, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 1, + "pipeline": "pl_test_foreach_coverage", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 8, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 8, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 21, + "complexity_size": "L", + "control_flow_activities": 2, + "coverage_pct": 100.0, + "databricks_native_activities": 2, + "datasets": 2, + "deterministic_activities": 8, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 1, + "migration_status": "included", + "other_activities": 4, + "pipeline": "pl_test_ifcondition_coverage", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 2, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 2, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 7, + "complexity_size": "M", + "control_flow_activities": 0, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 1, + "deterministic_activities": 2, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 2, + "pipeline": "pl_test_lookup_coverage", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 1, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 1, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 2, + "complexity_size": "S", + "control_flow_activities": 0, + "coverage_pct": 100.0, + "databricks_native_activities": 1, + "datasets": 0, + "deterministic_activities": 1, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 1, + "migration_status": "included", + "other_activities": 0, + "pipeline": "pl_test_notebook_coverage", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 5, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 5, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 10, + "complexity_size": "M", + "control_flow_activities": 5, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 0, + "deterministic_activities": 5, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 0, + "pipeline": "pl_test_setvariable_coverage", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 1, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 1, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 2, + "complexity_size": "S", + "control_flow_activities": 0, + "coverage_pct": 100.0, + "databricks_native_activities": 1, + "datasets": 0, + "deterministic_activities": 1, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 1, + "migration_status": "included", + "other_activities": 0, + "pipeline": "pl_test_sparkjar_coverage", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 1, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 1, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 2, + "complexity_size": "S", + "control_flow_activities": 0, + "coverage_pct": 100.0, + "databricks_native_activities": 1, + "datasets": 0, + "deterministic_activities": 1, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 1, + "migration_status": "included", + "other_activities": 0, + "pipeline": "pl_test_sparkpython_coverage", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 6, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 6, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 10, + "complexity_size": "M", + "control_flow_activities": 1, + "coverage_pct": 100.0, + "databricks_native_activities": 4, + "datasets": 0, + "deterministic_activities": 6, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 1, + "migration_status": "included", + "other_activities": 1, + "pipeline": "pl_test_switch_coverage", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 2, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 2, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 4, + "complexity_size": "S", + "control_flow_activities": 2, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 0, + "deterministic_activities": 2, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 0, + "pipeline": "pl_test_wait_coverage", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 3, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 3, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 9, + "complexity_size": "M", + "control_flow_activities": 0, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 0, + "deterministic_activities": 3, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 3, + "pipeline": "pl_test_webactivity_coverage", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + } +] \ No newline at end of file diff --git a/tests/unit/test_adf_dataset_lineage.py b/tests/unit/test_adf_dataset_lineage.py new file mode 100644 index 00000000..8d3f9a54 --- /dev/null +++ b/tests/unit/test_adf_dataset_lineage.py @@ -0,0 +1,239 @@ +"""Tests for the ADF dataset-lineage resolver (:mod:`flowx.sources.adf.dataset_lineage`). + +Proves the two-tier identity / signature model re-homed from the closed #36 +attempt: a table or a concrete path resolves to a physical ``identity``; a +parameterised reference does not (it never guesses) and falls back to the +structural path signature or the dataset name; and every input / output slot is +captured, not just index 0. +""" + +from __future__ import annotations + +from flowx.models.adf_ast import ( + AdfActivity, + AdfDataset, + AdfDatasetReference, + AdfDefinitions, + AdfLinkedService, +) +from flowx.sources.adf.dataset_lineage import activity_data_assets, resolve_dataset_identity + + +def _definitions(**datasets: AdfDataset) -> AdfDefinitions: + return AdfDefinitions(pipelines=[], datasets=dict(datasets)) + + +def _table_dataset(name: str, *, schema: str, table: str) -> AdfDataset: + return AdfDataset( + name=name, + type="AzureSqlTable", + properties={"typeProperties": {"schema": schema, "table": table}}, + ) + + +def _adls_dataset(name: str, *, file_system: str, folder_path: str, linked_service: str) -> AdfDataset: + return AdfDataset( + name=name, + type="DelimitedText", + properties={ + "typeProperties": {"location": {"fileSystem": file_system, "folderPath": folder_path}}, + "linkedServiceName": {"referenceName": linked_service}, + }, + ) + + +def _adls_linked_service(name: str, *, account: str) -> AdfLinkedService: + return AdfLinkedService( + name=name, + type="AzureBlobFS", + properties={"typeProperties": {"url": f"https://{account}.dfs.core.windows.net"}}, + ) + + +# --------------------------------------------------------------------------- # +# Identity tier +# --------------------------------------------------------------------------- # + + +def test_identity_resolves_schema_and_table() -> None: + definitions = _definitions(ds_orders=_table_dataset("ds_orders", schema="curated", table="orders")) + identity = resolve_dataset_identity(AdfDatasetReference(reference_name="ds_orders"), definitions) + assert identity == "curated.orders" + + +def test_identity_resolves_storage_path_from_linked_service() -> None: + definitions = AdfDefinitions( + pipelines=[], + datasets={ + "ds_raw": _adls_dataset("ds_raw", file_system="data", folder_path="raw/customers", linked_service="ls") + }, + linked_services={"ls": _adls_linked_service("ls", account="contosolake")}, + ) + identity = resolve_dataset_identity(AdfDatasetReference(reference_name="ds_raw"), definitions) + assert identity == "abfss://data@contosolake.dfs.core.windows.net/raw/customers" + + +def test_identity_resolves_dataset_param_from_call_site_literal() -> None: + """A ``@dataset().table`` expression resolves when the call site passes a literal.""" + dataset = AdfDataset( + name="ds_param", + type="AzureSqlTable", + properties={ + "typeProperties": {"schema": "dbo", "table": "@dataset().tbl"}, + "parameters": {"tbl": {"type": "String"}}, + }, + ) + definitions = _definitions(ds_param=dataset) + reference = AdfDatasetReference(reference_name="ds_param", parameters={"tbl": "shipments"}) + assert resolve_dataset_identity(reference, definitions) == "dbo.shipments" + + +def test_parameterised_table_is_not_guessed() -> None: + """An unresolved ``@pipeline()`` expression yields no identity (never a guess, per #36).""" + dataset = AdfDataset( + name="ds_dyn", + type="AzureSqlTable", + properties={"typeProperties": {"schema": "dbo", "table": "@pipeline().parameters.tableName"}}, + ) + definitions = _definitions(ds_dyn=dataset) + assert resolve_dataset_identity(AdfDatasetReference(reference_name="ds_dyn"), definitions) is None + + +def test_unknown_dataset_reference_has_no_identity() -> None: + assert resolve_dataset_identity(AdfDatasetReference(reference_name="missing"), _definitions()) is None + + +# --------------------------------------------------------------------------- # +# activity_data_assets: reads/writes + the two-tier signature +# --------------------------------------------------------------------------- # + + +def test_copy_captures_source_read_and_sink_write_with_identity() -> None: + definitions = _definitions( + ds_src=_table_dataset("ds_src", schema="raw", table="orders"), + ds_dst=_table_dataset("ds_dst", schema="curated", table="orders"), + ) + activity = AdfActivity( + name="Copy Orders", + type="Copy", + type_properties={ + "source": {"referenceName": "ds_src"}, + "sink": {"referenceName": "ds_dst"}, + }, + ) + reads, writes = activity_data_assets(activity, definitions) + assert [(asset.identity, asset.signature) for asset in reads] == [("raw.orders", "raw.orders")] + assert [(asset.identity, asset.signature) for asset in writes] == [("curated.orders", "curated.orders")] + assert reads[0].asset_type == "table" + + +def test_captures_all_inputs_and_outputs_not_just_index_zero() -> None: + """Every ``inputs`` / ``outputs`` slot is captured, not only the first.""" + definitions = _definitions( + ds_in_a=_table_dataset("ds_in_a", schema="raw", table="a"), + ds_in_b=_table_dataset("ds_in_b", schema="raw", table="b"), + ds_out_a=_table_dataset("ds_out_a", schema="curated", table="a"), + ds_out_b=_table_dataset("ds_out_b", schema="curated", table="b"), + ) + activity = AdfActivity( + name="Multi IO", + type="Copy", + inputs=[AdfDatasetReference(reference_name="ds_in_a"), AdfDatasetReference(reference_name="ds_in_b")], + outputs=[AdfDatasetReference(reference_name="ds_out_a"), AdfDatasetReference(reference_name="ds_out_b")], + ) + reads, writes = activity_data_assets(activity, definitions) + assert sorted(asset.identity for asset in reads) == ["raw.a", "raw.b"] + assert sorted(asset.identity for asset in writes) == ["curated.a", "curated.b"] + + +def test_dataset_named_in_both_slot_and_typeproperties_counted_once() -> None: + definitions = _definitions(ds_src=_table_dataset("ds_src", schema="raw", table="orders")) + activity = AdfActivity( + name="Lookup", + type="Lookup", + inputs=[AdfDatasetReference(reference_name="ds_src")], + type_properties={"dataset": {"referenceName": "ds_src"}}, + ) + reads, _ = activity_data_assets(activity, definitions) + assert [asset.identity for asset in reads] == ["raw.orders"] + + +def test_unresolved_reference_falls_back_to_path_signature() -> None: + """A parameterised file reference with a literal folder anchor gets a structural signature.""" + dataset = AdfDataset( + name="ds_wm", + type="DelimitedText", + properties={"typeProperties": {"location": {"fileSystem": "@dataset().fs"}}}, + ) + definitions = _definitions(ds_wm=dataset) + reference = AdfDatasetReference( + reference_name="ds_wm", + parameters={"folderPath": "@concat('watermark/', item().entity)", "fileName": "version.txt"}, + ) + activity = AdfActivity(name="Read WM", type="Lookup", type_properties={"dataset": _ref_dict(reference)}) + reads, _ = activity_data_assets(activity, definitions) + assert reads[0].identity is None + assert reads[0].signature == "FP[watermark|slots=1]/FN[version.txt|slots=0]" + + +def test_unresolved_reference_without_path_anchor_has_empty_signature() -> None: + """No identity and no path anchor -> empty signature, so it never joins (#36 name rule).""" + dataset = AdfDataset( + name="ds_generic", + type="AzureSqlTable", + properties={"typeProperties": {"table": "@pipeline().parameters.t"}}, + ) + definitions = _definitions(ds_generic=dataset) + activity = AdfActivity( + name="Copy", + type="Copy", + type_properties={"source": {"referenceName": "ds_generic"}}, + ) + reads, _ = activity_data_assets(activity, definitions) + assert reads[0].identity is None + # Never the bare dataset name: an empty signature cannot produce a signature-tier join. + assert reads[0].signature == "" + + +def test_distinct_parameter_bindings_of_same_dataset_are_all_retained() -> None: + """Same dataset ref, different params -> distinct physical assets, both kept (not collapsed).""" + dataset = AdfDataset( + name="ds", + type="AzureSqlTable", + properties={ + "typeProperties": {"schema": "raw", "table": "@dataset().tbl"}, + "parameters": {"tbl": {"type": "String"}}, + }, + ) + definitions = _definitions(ds=dataset) + activity = AdfActivity( + name="Multi Bind", + type="Copy", + inputs=[ + AdfDatasetReference(reference_name="ds", parameters={"tbl": "orders"}), + AdfDatasetReference(reference_name="ds", parameters={"tbl": "customers"}), + ], + ) + reads, _ = activity_data_assets(activity, definitions) + assert sorted(asset.identity for asset in reads) == ["raw.customers", "raw.orders"] + + +def test_same_ref_same_params_in_slot_and_typeproperties_still_collapses() -> None: + """The de-dup only fires on an identical binding: one asset, not two.""" + definitions = _definitions(ds_src=_table_dataset("ds_src", schema="raw", table="orders")) + activity = AdfActivity( + name="Lookup", + type="Lookup", + inputs=[AdfDatasetReference(reference_name="ds_src")], + type_properties={"dataset": {"referenceName": "ds_src"}}, + ) + reads, _ = activity_data_assets(activity, definitions) + assert [asset.identity for asset in reads] == ["raw.orders"] + + +def _ref_dict(reference: AdfDatasetReference) -> dict: + """Render a dataset reference the way ADF ``typeProperties`` nests it.""" + payload: dict = {"referenceName": reference.reference_name} + if reference.parameters: + payload["parameters"] = reference.parameters + return payload diff --git a/tests/unit/test_adf_discovery_mapping.py b/tests/unit/test_adf_discovery_mapping.py new file mode 100644 index 00000000..71d72b3b --- /dev/null +++ b/tests/unit/test_adf_discovery_mapping.py @@ -0,0 +1,423 @@ +"""Tests for the ADF -> shared discovery AST mapper (:mod:`flowx.sources.adf.discovery_mapping`). + +Proves the mapping is 1:1 and lossless: every activity becomes one discovery +node, the ADF type is retained verbatim as ``native_type`` / ``original_type``, +the source dict is preserved on ``raw``, dependency edges keep all of their +outcome conditions, and control-flow nesting maps to labelled container branches. +""" + +from __future__ import annotations + +from flowx.discovery_inventory import STRATEGY_PROPERTY +from flowx.models.adf_ast import AdfDefinitions +from flowx.models.discovery import ( + SOURCE_ADF, + ContainerNode, + GapNode, + SourceGraph, + SourceNode, +) +from flowx.sources.adf.discovery_mapping import ( + adf_definitions_to_source_graphs, + adf_pipeline_to_source_graph, +) +from flowx.sources.adf.loader import _parse_pipeline_json, _parse_trigger_json + + +def _pipeline(activities: list[dict], **props) -> SourceGraph: + data = {"name": props.pop("name", "pl"), "properties": {"activities": activities, **props}} + return adf_pipeline_to_source_graph(_parse_pipeline_json(data)) + + +def test_activity_maps_1to1_retaining_type_and_raw() -> None: + """Each activity becomes one node with its ADF type and raw dict preserved.""" + raw_activity = { + "name": "Run Notebook", + "type": "DatabricksNotebook", + "typeProperties": {"notebookPath": "/nb"}, + } + graph = _pipeline([raw_activity]) + + assert len(graph.tasks) == 1 + node = graph.tasks[0] + assert isinstance(node, SourceNode) + assert node.source == SOURCE_ADF + assert node.name == "Run Notebook" + assert node.task_key == "Run Notebook" + # ADF type is retained verbatim -- native_type is the original_type source. + assert node.native_type == "DatabricksNotebook" + # Verbatim source dict preserved for lossless fallback. + assert node.raw == raw_activity + # Target-side strategy stashed in the properties seam (not a typed field). + assert node.properties[STRATEGY_PROPERTY] == "deterministic" + + +def test_dependencies_capture_all_conditions() -> None: + """A dependency edge keeps every outcome condition, not just the first.""" + activities = [ + {"name": "A", "type": "Wait", "typeProperties": {"waitTimeInSeconds": 1}}, + { + "name": "B", + "type": "DatabricksNotebook", + "dependsOn": [{"activity": "A", "dependencyConditions": ["Succeeded", "Skipped"]}], + }, + ] + graph = _pipeline(activities) + + node_b = next(node for node in graph.tasks if node.name == "B") + assert len(node_b.dependencies) == 1 + assert node_b.dependencies[0].upstream == "A" + assert node_b.dependencies[0].conditions == ["Succeeded", "Skipped"] + + +def test_no_motif_collapse_every_activity_is_a_node() -> None: + """Activities that a motif would merge each stay their own node (no collapse).""" + activities = [ + {"name": "Load", "type": "Copy"}, + {"name": "Notify", "type": "WebActivity", "dependsOn": [{"activity": "Load"}]}, + ] + graph = _pipeline(activities) + assert [node.name for node in graph.tasks] == ["Load", "Notify"] + + +def test_control_flow_maps_to_container_branches() -> None: + """An IfCondition maps to a ContainerNode with true/false branches nested.""" + activities = [ + { + "name": "Check", + "type": "IfCondition", + "typeProperties": { + "ifTrueActivities": [{"name": "T", "type": "DatabricksNotebook"}], + "ifFalseActivities": [{"name": "F", "type": "Wait"}], + }, + } + ] + graph = _pipeline(activities) + + container = graph.tasks[0] + assert isinstance(container, ContainerNode) + assert container.native_type == "IfCondition" + assert list(container.branches.keys()) == ["true", "false"] + assert container.branches["true"][0].name == "T" + assert container.branches["false"][0].name == "F" + + +def test_switch_maps_cases_and_default_to_branches() -> None: + """A Switch maps each case value plus default to its own labelled branch.""" + activities = [ + { + "name": "Route", + "type": "Switch", + "typeProperties": { + "cases": [ + {"value": "gold", "activities": [{"name": "G", "type": "Copy"}]}, + {"value": "silver", "activities": [{"name": "S", "type": "Copy"}]}, + ], + "defaultActivities": [{"name": "D", "type": "Wait"}], + }, + } + ] + graph = _pipeline(activities) + + container = graph.tasks[0] + assert isinstance(container, ContainerNode) + assert list(container.branches.keys()) == ["gold", "silver", "default"] + assert container.branches["gold"][0].name == "G" + assert container.branches["default"][0].name == "D" + + +def test_unsupported_activity_becomes_gap_node() -> None: + """An unsupported ADF type maps to a GapNode carrying the reason and raw.""" + activities = [{"name": "Weird", "type": "TotallyUnknownType"}] + graph = _pipeline(activities) + + node = graph.tasks[0] + assert isinstance(node, GapNode) + assert node.reason is not None and "TotallyUnknownType" in node.reason + assert node.raw == {"name": "Weird", "type": "TotallyUnknownType"} + assert node.properties[STRATEGY_PROPERTY] == "unsupported" + + +def test_policy_maps_retry_and_preserves_timeout_verbatim() -> None: + """Retry count/interval map to typed fields; the ISO timeout rides in extensions.""" + activities = [ + { + "name": "Copy", + "type": "Copy", + "policy": { + "timeout": "0.12:00:00", + "retry": 3, + "retryIntervalInSeconds": 30, + "secureInput": True, + }, + } + ] + graph = _pipeline(activities) + + policy = graph.tasks[0].policy + assert policy is not None + assert policy.max_retries == 3 + assert policy.retry_interval_seconds == 30 + # Timeout normalisation to seconds is a target concern -> preserved verbatim. + assert policy.extensions["timeout"] == "0.12:00:00" + assert policy.extensions["secure_input"] is True + + +def test_graph_carries_parameters_variables_tags_and_folder() -> None: + """Graph-level metadata maps onto the shared fields / properties seam.""" + data = { + "name": "pl", + "properties": { + "activities": [{"name": "N", "type": "DatabricksNotebook"}], + "parameters": {"env": {"type": "String", "defaultValue": "dev"}}, + "variables": {"counter": {"type": "Integer"}}, + "annotations": ["team-a", "prod"], + "folder": {"name": "ingest/bronze"}, + }, + } + graph = adf_pipeline_to_source_graph(_parse_pipeline_json(data)) + + assert graph.source == SOURCE_ADF + assert graph.parameters["env"].type == "String" + assert graph.parameters["env"].default == "dev" + assert graph.variables["counter"].type == "Integer" + assert graph.tags == ["team-a", "prod"] + assert graph.properties["folder"] == "ingest/bronze" + assert graph.raw is not None # verbatim pipeline dict preserved + + +def test_definitions_map_preserves_pipeline_order(adf_definitions) -> None: + """The definitions-level mapper yields one graph per pipeline, in order.""" + graphs = adf_definitions_to_source_graphs(adf_definitions) + assert [g.name for g in graphs] == [p.name for p in adf_definitions.pipelines] + assert all(g.source == SOURCE_ADF for g in graphs) + + +# --------------------------------------------------------------------------- +# Triggers / schedules (BLOCKING 1) +# --------------------------------------------------------------------------- + + +def _definitions_with_trigger(trigger: dict) -> AdfDefinitions: + pipeline = _parse_pipeline_json({"name": "pl_sched", "properties": {"activities": []}}) + return AdfDefinitions(pipelines=[pipeline], triggers=[_parse_trigger_json(trigger)]) + + +def test_schedule_trigger_lands_in_source_graph_schedule() -> None: + """A ScheduleTrigger referencing a pipeline populates that graph's schedule.""" + trigger = { + "name": "tr_daily", + "properties": { + "type": "ScheduleTrigger", + "typeProperties": { + "recurrence": {"frequency": "Day", "interval": 1, "timeZone": "UTC"}, + }, + "pipelines": [{"pipelineReference": {"referenceName": "pl_sched", "type": "PipelineReference"}}], + }, + } + graphs = adf_definitions_to_source_graphs(_definitions_with_trigger(trigger)) + + schedule = graphs[0].schedule + assert schedule is not None + assert schedule.kind == "schedule" + # Recurrence payload preserved verbatim as the expression. + assert schedule.expression == {"frequency": "Day", "interval": 1, "timeZone": "UTC"} + assert schedule.timezone == "UTC" + # Full trigger properties preserved losslessly in extensions. + assert schedule.extensions["trigger_name"] == "tr_daily" + assert schedule.extensions["trigger_type"] == "ScheduleTrigger" + assert "properties" in schedule.extensions + + +def test_tumbling_window_trigger_expression_from_type_properties() -> None: + """A TumblingWindowTrigger keeps its typeProperties as the expression.""" + trigger = { + "name": "tr_tumble", + "properties": { + "type": "TumblingWindowTrigger", + "typeProperties": {"frequency": "Hour", "interval": 1, "startTime": "2024-01-01T00:00:00Z"}, + "pipelines": [{"pipelineReference": {"referenceName": "pl_sched"}}], + }, + } + graphs = adf_definitions_to_source_graphs(_definitions_with_trigger(trigger)) + + schedule = graphs[0].schedule + assert schedule is not None + assert schedule.kind == "interval" + assert schedule.expression == {"frequency": "Hour", "interval": 1, "startTime": "2024-01-01T00:00:00Z"} + + +def test_unreferenced_pipeline_has_no_schedule() -> None: + """A pipeline no trigger references keeps ``schedule is None``.""" + trigger = { + "name": "tr_other", + "properties": { + "type": "ScheduleTrigger", + "typeProperties": {"recurrence": {"frequency": "Day", "interval": 1}}, + "pipelines": [{"pipelineReference": {"referenceName": "some_other_pipeline"}}], + }, + } + graphs = adf_definitions_to_source_graphs(_definitions_with_trigger(trigger)) + assert graphs[0].schedule is None + + +def test_multiple_triggers_preserve_extras_in_extensions() -> None: + """A second trigger for the same pipeline is preserved, not overwritten.""" + pipeline = _parse_pipeline_json({"name": "pl_sched", "properties": {"activities": []}}) + triggers = [ + _parse_trigger_json( + { + "name": "tr_first", + "properties": { + "type": "ScheduleTrigger", + "typeProperties": {"recurrence": {"frequency": "Day", "interval": 1}}, + "pipelines": [{"pipelineReference": {"referenceName": "pl_sched"}}], + }, + } + ), + _parse_trigger_json( + { + "name": "tr_second", + "properties": { + "type": "ScheduleTrigger", + "typeProperties": {"recurrence": {"frequency": "Hour", "interval": 6}}, + "pipelines": [{"pipelineReference": {"referenceName": "pl_sched"}}], + }, + } + ), + ] + graphs = adf_definitions_to_source_graphs(AdfDefinitions(pipelines=[pipeline], triggers=triggers)) + + schedule = graphs[0].schedule + assert schedule is not None + assert schedule.extensions["trigger_name"] == "tr_first" # first wins the typed slot + additional = schedule.extensions["additional_triggers"] + assert len(additional) == 1 + # The additional trigger retains its NAME (not just properties). + assert additional[0]["trigger_name"] == "tr_second" + assert additional[0]["trigger_type"] == "ScheduleTrigger" + assert additional[0]["properties"]["typeProperties"]["recurrence"]["interval"] == 6 + + +def test_triggers_do_not_leak_across_pipelines() -> None: + """A ScheduleSpec is per-pipeline: a later A-only trigger must not appear on B. + + Guards the shared-instance aliasing bug -- trigger_ab references A and B, then + trigger_a references only A. B must keep exactly trigger_ab and gain nothing + from trigger_a. + """ + pipeline_a = _parse_pipeline_json({"name": "pl_a", "properties": {"activities": []}}) + pipeline_b = _parse_pipeline_json({"name": "pl_b", "properties": {"activities": []}}) + triggers = [ + _parse_trigger_json( + { + "name": "tr_ab", + "properties": { + "type": "ScheduleTrigger", + "typeProperties": {"recurrence": {"frequency": "Day", "interval": 1}}, + "pipelines": [ + {"pipelineReference": {"referenceName": "pl_a"}}, + {"pipelineReference": {"referenceName": "pl_b"}}, + ], + }, + } + ), + _parse_trigger_json( + { + "name": "tr_a_only", + "properties": { + "type": "ScheduleTrigger", + "typeProperties": {"recurrence": {"frequency": "Hour", "interval": 2}}, + "pipelines": [{"pipelineReference": {"referenceName": "pl_a"}}], + }, + } + ), + ] + graphs = { + graph.name: graph + for graph in adf_definitions_to_source_graphs( + AdfDefinitions(pipelines=[pipeline_a, pipeline_b], triggers=triggers) + ) + } + + schedule_a = graphs["pl_a"].schedule + schedule_b = graphs["pl_b"].schedule + assert schedule_a is not None and schedule_b is not None + assert schedule_a is not schedule_b # distinct instances, no aliasing + # A picked up the second trigger; B must NOT have leaked it. + assert schedule_a.extensions["additional_triggers"][0]["trigger_name"] == "tr_a_only" + assert "additional_triggers" not in schedule_b.extensions + + +def test_fixture_scheduled_pipeline_gets_schedule(adf_definitions) -> None: + """End-to-end over the fixtures: a trigger-referenced pipeline gets a schedule.""" + graphs = {graph.name: graph for graph in adf_definitions_to_source_graphs(adf_definitions)} + # tr_daily_schedule references pipeline_copy_csv_to_delta in the fixtures. + assert graphs["pipeline_copy_csv_to_delta"].schedule is not None + + +# --------------------------------------------------------------------------- +# Empty / one-sided control flow (BLOCKING 2) +# --------------------------------------------------------------------------- + + +def test_empty_if_condition_stays_a_container_with_both_branches() -> None: + """An IfCondition with no children still maps to a ContainerNode, both branches present.""" + graph = _pipeline([{"name": "Gate", "type": "IfCondition", "typeProperties": {}}]) + + node = graph.tasks[0] + assert isinstance(node, ContainerNode) + assert node.native_type == "IfCondition" + assert list(node.branches.keys()) == ["true", "false"] + assert node.branches["true"] == [] + assert node.branches["false"] == [] + + +def test_empty_for_each_stays_a_container_with_body_branch() -> None: + """A ForEach with no children still maps to a ContainerNode with an empty body.""" + graph = _pipeline([{"name": "Loop", "type": "ForEach", "typeProperties": {}}]) + + node = graph.tasks[0] + assert isinstance(node, ContainerNode) + assert list(node.branches.keys()) == ["body"] + assert node.branches["body"] == [] + + +def test_empty_until_stays_a_container_with_body_branch() -> None: + """An Until with no children still maps to a ContainerNode with an empty body.""" + graph = _pipeline([{"name": "Retry", "type": "Until", "typeProperties": {}}]) + + node = graph.tasks[0] + assert isinstance(node, ContainerNode) + assert node.native_type == "Until" + assert list(node.branches.keys()) == ["body"] + assert node.branches["body"] == [] + + +def test_one_sided_if_keeps_empty_false_branch() -> None: + """An If with only a true branch keeps its false branch present-but-empty.""" + graph = _pipeline( + [ + { + "name": "Gate", + "type": "IfCondition", + "typeProperties": {"ifTrueActivities": [{"name": "T", "type": "Wait"}]}, + } + ] + ) + + node = graph.tasks[0] + assert isinstance(node, ContainerNode) + assert [child.name for child in node.branches["true"]] == ["T"] + assert "false" in node.branches # empty branch is present, not dropped + assert node.branches["false"] == [] + + +def test_empty_switch_stays_a_container_with_default_branch() -> None: + """A Switch with no cases still maps to a ContainerNode with an empty default.""" + graph = _pipeline([{"name": "Route", "type": "Switch", "typeProperties": {}}]) + + node = graph.tasks[0] + assert isinstance(node, ContainerNode) + assert list(node.branches.keys()) == ["default"] + assert node.branches["default"] == [] diff --git a/tests/unit/test_adf_inventory_superset.py b/tests/unit/test_adf_inventory_superset.py new file mode 100644 index 00000000..8963cece --- /dev/null +++ b/tests/unit/test_adf_inventory_superset.py @@ -0,0 +1,193 @@ +"""Consumer-safety tests for the ADF inventory emitted via the shared AST + emitter. + +Two guarantees: + +* **Superset** -- the new ``inventory.json`` keeps every key the historical shape + had (per-activity ``name`` / ``type`` / ``strategy`` / ``depends_on`` and the + ``summary`` count block) byte-compatibly, and only *adds* fields on top. The + historical values are reconstructed here from :func:`build_inventory` (the + authoritative classifier, untouched by this slice). +* **Golden coverage** -- feeding the new inventory through the real consumer + (:func:`flowx.reporting.coverage.build_coverage_rows`) reproduces a committed + snapshot, so downstream reporting is provably unchanged. +""" + +from __future__ import annotations + +import json +from pathlib import Path + +from flowx.reporting.coverage import build_coverage_rows +from flowx.sources.adf.loader import build_inventory, load_adf_definitions, main + +FIXTURES_DIR = Path(__file__).parent.parent / "resources" / "json" +GOLDEN_COVERAGE = Path(__file__).parent.parent / "resources" / "golden" / "adf_fixture_coverage.json" + + +def _run_discover(tmp_path: Path) -> Path: + """Run the ADF discover entry point against the fixtures; return metadata dir.""" + exit_code = main(["--source-dir", str(FIXTURES_DIR), "--output-dir", str(tmp_path)]) + assert exit_code == 0 + return tmp_path / "metadata" + + +def _legacy_pipeline_map() -> dict[str, list[dict]]: + """Reconstruct the pre-change per-pipeline activity shape from build_inventory. + + This is exactly what the removed ``_inventory_to_dict`` used to emit, rebuilt + from the authoritative classifier so the superset check does not depend on a + frozen copy of the old serializer. + """ + inventory = build_inventory(load_adf_definitions(FIXTURES_DIR)) + pipeline_map: dict[str, list[dict]] = {} + for item in inventory.items: + entry: dict = {"name": item.activity_name, "type": item.activity_type, "strategy": item.strategy.value} + if item.depends_on: + entry["depends_on"] = item.depends_on + pipeline_map.setdefault(item.pipeline_name, []).append(entry) + return pipeline_map + + +def test_inventory_top_level_is_superset(tmp_path: Path) -> None: + """Top-level keys include the legacy set plus the new ``source`` discriminator.""" + metadata = _run_discover(tmp_path) + inventory = json.loads((metadata / "inventory.json").read_text()) + + # Legacy top-level keys still present. + for key in ("source_dir", "pipelines", "summary"): + assert key in inventory + # Additive discriminator. + assert inventory["source"] == "adf" + + +def test_summary_counts_match_legacy_classifier(tmp_path: Path) -> None: + """The summary count block is byte-compatible with the legacy classifier.""" + metadata = _run_discover(tmp_path) + summary = json.loads((metadata / "inventory.json").read_text())["summary"] + + legacy = build_inventory(load_adf_definitions(FIXTURES_DIR)) + total = legacy.deterministic_count + legacy.agentic_count + legacy.unsupported_count + assert summary["pipeline_count"] == legacy.pipeline_count + assert summary["activity_count"] == total + assert summary["deterministic_count"] == legacy.deterministic_count + assert summary["agentic_count"] == legacy.agentic_count + assert summary["unsupported_count"] == legacy.unsupported_count + assert summary["coverage_pct"] == round((legacy.deterministic_count + legacy.agentic_count) / total * 100, 1) + + +def test_activity_entries_superset_legacy_shape(tmp_path: Path) -> None: + """Every activity keeps its legacy keys/values and only gains additive fields.""" + metadata = _run_discover(tmp_path) + inventory = json.loads((metadata / "inventory.json").read_text()) + legacy_map = _legacy_pipeline_map() + + # Same pipeline membership (ADF omits zero-activity pipelines; fixtures have none empty). + new_names = [pipeline["name"] for pipeline in inventory["pipelines"]] + assert sorted(new_names) == sorted(legacy_map) + + for pipeline in inventory["pipelines"]: + legacy_entries = legacy_map[pipeline["name"]] + new_entries = pipeline["activities"] + assert len(new_entries) == len(legacy_entries) + for legacy_entry, new_entry in zip(legacy_entries, new_entries): + # Every legacy key/value survives byte-for-byte. + for key, value in legacy_entry.items(): + assert new_entry[key] == value, (pipeline["name"], key) + # Additive standardized fields are present. + assert new_entry["original_type"] == legacy_entry["type"] + assert "dependencies" in new_entry + assert "raw" in new_entry + + +def test_dependencies_field_carries_conditions(tmp_path: Path) -> None: + """The additive ``dependencies`` field carries upstream + conditions per edge.""" + metadata = _run_discover(tmp_path) + inventory = json.loads((metadata / "inventory.json").read_text()) + + # The all-dependency-conditions fixture exercises non-default conditions. + pipeline = next(p for p in inventory["pipelines"] if p["name"] == "pipeline_all_dependency_conditions") + edges = [dependency for activity in pipeline["activities"] for dependency in activity["dependencies"]] + assert edges, "expected at least one dependency edge" + for edge in edges: + assert set(edge.keys()) == {"upstream", "conditions", "resolved"} + assert isinstance(edge["conditions"], list) + + +def test_inventory_carries_per_pipeline_lineage(tmp_path: Path) -> None: + """The emitted inventory surfaces the lineage #62b derived over the ADF fixtures. + + Lineage rides additively on each pipeline entry, so the aggregate across the + per-pipeline blocks must match what the discovery lineage pass computed: on + these fixtures that is 11 cross-pipeline control edges and 0 data edges. + """ + metadata = _run_discover(tmp_path) + inventory = json.loads((metadata / "inventory.json").read_text()) + + control_edges = 0 + data_edges = 0 + for pipeline in inventory["pipelines"]: + assert "lineage" in pipeline, pipeline["name"] + # Additive block only -- historical per-pipeline keys are untouched. + assert set(pipeline["lineage"].keys()) == {"control_edges", "data_edges", "motifs"} + control_edges += len(pipeline["lineage"]["control_edges"]) + data_edges += len(pipeline["lineage"]["data_edges"]) + + assert control_edges == 11 + assert data_edges == 0 + + +def test_inventory_surfaces_detected_motifs_without_collapsing(tmp_path: Path) -> None: + """Discover surfaces the profiler's detected motifs additively, members intact. + + ``pipeline_complex_etl`` carries a detectable metadata-driven bulk-copy motif. + It must appear in the pipeline's additive ``motifs`` list -- carrying its type, + Databricks replacement target, and participating activities -- while every + member still appears as its own entry in ``activities`` (no discover-time + collapse) and the lineage block's own motif slot stays empty (decoupled). + """ + metadata = _run_discover(tmp_path) + inventory = json.loads((metadata / "inventory.json").read_text()) + + pipeline = next(p for p in inventory["pipelines"] if p["name"] == "pipeline_complex_etl") + bulk = next(m for m in pipeline["motifs"] if m["motif_id"] == "metadata_driven_bulk_copy") + + assert set(bulk.keys()) == { + "motif_id", + "display_name", + "databricks_replacement", + "member_task_keys", + "source_type_hint", + "confidence_notes", + } + assert bulk["databricks_replacement"] == "for_each_ingestion" + # Exact expected member set for this fixture -- a partial-member regression must fail. + assert bulk["member_task_keys"] == ["Lookup ETL Config", "ForEach Table In Config"] + + # No collapse at discover: each member survives as its own activity entry. + activity_names = {activity["name"] for activity in pipeline["activities"]} + for member in bulk["member_task_keys"]: + assert member in activity_names, member + + # Decoupled from lineage: the additive motifs key is separate from lineage.motifs. + assert pipeline["lineage"]["motifs"] == [] + + +def test_pipelines_without_a_motif_omit_the_motifs_key(tmp_path: Path) -> None: + """The motifs field is additive: a pipeline with no detected motif omits it.""" + metadata = _run_discover(tmp_path) + inventory = json.loads((metadata / "inventory.json").read_text()) + + plain = next(p for p in inventory["pipelines"] if p["name"] == "pipeline_notebook_basic") + assert "motifs" not in plain + + +def test_coverage_output_matches_golden(tmp_path: Path) -> None: + """The real coverage consumer reproduces the committed golden snapshot. + + This is the no-consumer-breakage guarantee: regenerate the golden with + ``make test`` after an intentional change and review the diff. + """ + metadata = _run_discover(tmp_path) + rows = json.loads(json.dumps(build_coverage_rows(metadata), sort_keys=True)) + golden = json.loads(GOLDEN_COVERAGE.read_text()) + assert rows == golden diff --git a/tests/unit/test_discovery_inventory.py b/tests/unit/test_discovery_inventory.py new file mode 100644 index 00000000..3e5b21e4 --- /dev/null +++ b/tests/unit/test_discovery_inventory.py @@ -0,0 +1,361 @@ +"""Tests for the source-agnostic inventory emitter (:mod:`flowx.discovery_inventory`). + +These tests build the shared discovery AST by hand -- no ADF, no Airflow -- so +they prove the emitter is genuinely source-agnostic: it takes ``SourceGraph`` +objects in and projects the ``inventory.json`` document out, with no coupling to +any particular front-end. +""" + +from __future__ import annotations + +import ast + +import flowx.discovery_inventory as discovery_inventory +from flowx.discovery_inventory import STRATEGY_PROPERTY, build_source_inventory +from flowx.discovery_serde import source_graph_from_dict, source_graph_to_dict +from flowx.models.discovery import ( + CONCEPT_BRANCH, + CONCEPT_NOTEBOOK, + ContainerNode, + SourceDependency, + SourceGraph, + SourceNode, +) +from flowx.models.ir import ControlEdge, DataEdge, Lineage +from flowx.models.motifs import DetectedMotif, MotifDefinition + + +def _motif( + motif_id: str, + replacement: str, + members: list[str], + *, + hint: str | None = None, + notes: list[str] | None = None, +) -> DetectedMotif: + definition = MotifDefinition( + motif_id=motif_id, + display_name=motif_id, + description="", + expected_activity_types=(), + databricks_replacement=replacement, + ) + return DetectedMotif( + definition=definition, + matched_activities=list(members), + source_type_hint=hint, + confidence_notes=list(notes or []), + ) + + +def _node(task_key: str, native_type: str, strategy: str, *, deps: list[SourceDependency] | None = None) -> SourceNode: + return SourceNode( + source_id=task_key, + task_key=task_key, + concept=CONCEPT_NOTEBOOK, + source="unit", + name=task_key, + native_type=native_type, + dependencies=deps or [], + properties={STRATEGY_PROPERTY: strategy}, + raw={"name": task_key, "type": native_type}, + ) + + +def test_emitter_has_no_source_specific_imports() -> None: + """The emitter module must not import any per-source package. + + Source-agnostic means the ADF/Airflow loaders depend on the emitter, never + the other way round. Guard that by inspecting the module's actual import + statements (not arbitrary text -- the docstring legitimately names the + ``"adf"`` / ``"airflow"`` discriminator values). + """ + source = (discovery_inventory.__file__ or "").rstrip("c") + with open(source, encoding="utf-8") as handle: + tree = ast.parse(handle.read()) + + imported: list[str] = [] + for node in ast.walk(tree): + if isinstance(node, ast.Import): + imported.extend(alias.name for alias in node.names) + elif isinstance(node, ast.ImportFrom) and node.module: + imported.append(node.module) + + assert not any(name.startswith("flowx.sources") for name in imported), imported + + +def test_top_level_shape_and_summary_counts() -> None: + """A hand-built graph projects to the canonical top-level shape and counts.""" + graph = SourceGraph( + name="g1", + source="unit", + tasks=[ + _node("n1", "Notebook", "deterministic"), + _node("n2", "DataFlow", "agentic"), + _node("n3", "Mystery", "unsupported"), + ], + ) + + inventory = build_source_inventory([graph], source="unit", source_dir="/tmp/src") + + assert sorted(inventory.keys()) == ["pipelines", "source", "source_dir", "summary"] + assert inventory["source"] == "unit" + assert inventory["source_dir"] == "/tmp/src" + assert inventory["summary"] == { + "pipeline_count": 1, + "activity_count": 3, + "deterministic_count": 1, + "agentic_count": 1, + "unsupported_count": 1, + "coverage_pct": 66.7, + } + + +def test_activity_entry_carries_legacy_and_additive_fields() -> None: + """Each activity keeps the legacy keys and gains the additive standardized ones.""" + graph = SourceGraph( + name="g1", + source="unit", + tasks=[ + _node("a", "Notebook", "deterministic"), + _node( + "b", + "Copy", + "deterministic", + deps=[SourceDependency(upstream="a", conditions=["Succeeded", "Skipped"])], + ), + ], + ) + + inventory = build_source_inventory([graph], source="unit", source_dir="/tmp/src") + entries = {entry["name"]: entry for entry in inventory["pipelines"][0]["activities"]} + + # Legacy keys (byte-compatible with the historical shape). + assert entries["a"]["type"] == "Notebook" + assert entries["a"]["strategy"] == "deterministic" + assert "depends_on" not in entries["a"] # no deps -> key omitted, as before + assert entries["b"]["depends_on"] == ["a"] + + # Additive standardized fields. + assert entries["a"]["original_type"] == "Notebook" + assert entries["a"]["dependencies"] == [] + assert entries["a"]["raw"] == {"name": "a", "type": "Notebook"} + assert entries["b"]["dependencies"] == [{"upstream": "a", "conditions": ["Succeeded", "Skipped"], "resolved": True}] + + +def test_container_branches_are_flattened_in_source_order() -> None: + """Container children are flattened depth-first in branch declaration order.""" + branch_true = _node("t", "Notebook", "deterministic") + branch_false = _node("f", "Notebook", "deterministic") + container = ContainerNode( + source_id="if", + task_key="if", + concept=CONCEPT_BRANCH, + source="unit", + name="if", + native_type="IfCondition", + properties={STRATEGY_PROPERTY: "deterministic"}, + branches={"true": [branch_true], "false": [branch_false]}, + ) + graph = SourceGraph(name="g", source="unit", tasks=[container, _node("after", "Notebook", "deterministic")]) + + inventory = build_source_inventory([graph], source="unit", source_dir="/tmp/src") + names = [entry["name"] for entry in inventory["pipelines"][0]["activities"]] + + assert names == ["if", "t", "f", "after"] + assert inventory["summary"]["activity_count"] == 4 + + +def test_include_empty_pipelines_toggle_preserves_membership_semantics() -> None: + """Zero-activity graphs stay counted in summary but drop from the listing when asked. + + Reproduces ADF's long-standing rule: a pipeline with no activities is omitted + from ``pipelines`` yet still counted in ``summary.pipeline_count``. + """ + empty = SourceGraph(name="empty", source="unit", tasks=[]) + populated = SourceGraph(name="full", source="unit", tasks=[_node("n", "Notebook", "deterministic")]) + + omitted = build_source_inventory( + [empty, populated], source="unit", source_dir="/tmp", include_empty_pipelines=False + ) + assert [p["name"] for p in omitted["pipelines"]] == ["full"] + assert omitted["summary"]["pipeline_count"] == 2 # empty still counted + + listed = build_source_inventory([empty, populated], source="unit", source_dir="/tmp", include_empty_pipelines=True) + assert [p["name"] for p in listed["pipelines"]] == ["empty", "full"] + assert listed["summary"]["pipeline_count"] == 2 + + +def test_empty_input_yields_zero_coverage() -> None: + """No graphs -> empty listing, zeroed summary, 0.0 coverage (no divide-by-zero).""" + inventory = build_source_inventory([], source="unit", source_dir="/tmp") + assert inventory["pipelines"] == [] + assert inventory["summary"]["coverage_pct"] == 0.0 + assert inventory["summary"]["pipeline_count"] == 0 + + +def test_pipeline_carries_derived_lineage_block_that_round_trips() -> None: + """A graph with derived lineage emits a per-pipeline block via the shared serialiser. + + The emitted block must be byte-identical to what the shared discovery serde + produces for the same graph, and it must rehydrate through that serde back to + the original :class:`Lineage` -- proving the emitter consumes the one shared + lineage serialisation rather than a second hand-rolled one. + """ + lineage = Lineage( + control_edges=[ + ControlEdge(source_workflow="g", target_workflow="child", via_task_key="call", wait_for_completion=True) + ], + data_edges=[DataEdge(source_task_key="a", target_task_key="b", match_kind="identity", match_key="cat.sch.tbl")], + ) + graph = SourceGraph( + name="g", + source="unit", + tasks=[_node("a", "Notebook", "deterministic")], + lineage=lineage, + ) + + inventory = build_source_inventory([graph], source="unit", source_dir="/tmp/src") + pipeline_entry = inventory["pipelines"][0] + + # The block is present and byte-identical to the shared serde's per-graph output. + assert pipeline_entry["lineage"] == source_graph_to_dict(graph)["lineage"] + + # It round-trips through the shared serde back to the original Lineage. + rehydrated = source_graph_from_dict( + {"name": graph.name, "source": graph.source, "lineage": pipeline_entry["lineage"]} + ) + assert rehydrated.lineage == lineage + + +def test_pipeline_without_lineage_omits_the_key() -> None: + """A graph with no derived lineage omits the additive key -- historical keys untouched. + + Backward-compat guard: the lineage key is additive-only, so a graph that never + had lineage derived leaves the pipeline entry exactly as before. + """ + graph = SourceGraph(name="g", source="unit", tasks=[_node("a", "Notebook", "deterministic")]) + + inventory = build_source_inventory([graph], source="unit", source_dir="/tmp/src") + + assert graph.lineage is None + assert "lineage" not in inventory["pipelines"][0] + assert sorted(inventory["pipelines"][0].keys()) == ["activities", "name"] + + +def test_detected_motifs_surface_additively_without_collapsing_members() -> None: + """A supplied motif projects to an additive per-pipeline ``motifs`` entry. + + The member activities are surfaced by key but never merged away -- each still + appears as its own entry in ``activities``, proving discover does not collapse. + """ + graph = SourceGraph( + name="g1", + source="unit", + tasks=[ + _node("load", "Copy", "deterministic"), + _node("notify", "WebActivity", "deterministic"), + ], + ) + motif = _motif( + "activity_and_notify", + "task_with_notification", + ["load", "notify"], + hint="database", + notes=["'notify' looks like a notification call"], + ) + + inventory = build_source_inventory([graph], source="unit", source_dir="/tmp", motifs_by_pipeline={"g1": [motif]}) + entry = inventory["pipelines"][0] + + assert entry["motifs"] == [ + { + "motif_id": "activity_and_notify", + "display_name": "activity_and_notify", + "databricks_replacement": "task_with_notification", + "member_task_keys": ["load", "notify"], + "source_type_hint": "database", + "confidence_notes": ["'notify' looks like a notification call"], + } + ] + # No collapse: both members remain their own activity entries. + assert [activity["name"] for activity in entry["activities"]] == ["load", "notify"] + + +def test_motifs_key_is_additive_and_omitted_when_none_detected() -> None: + """The ``motifs`` key only appears when a pipeline has a detected motif. + + A pipeline mapped to an empty list, or absent from the map entirely, keeps the + historical per-pipeline keys untouched -- the key is additive-only. + """ + graph = SourceGraph(name="g1", source="unit", tasks=[_node("a", "Notebook", "deterministic")]) + + empty = build_source_inventory([graph], source="unit", source_dir="/tmp", motifs_by_pipeline={"g1": []}) + assert "motifs" not in empty["pipelines"][0] + + unmapped = build_source_inventory([graph], source="unit", source_dir="/tmp", motifs_by_pipeline=None) + assert "motifs" not in unmapped["pipelines"][0] + assert sorted(unmapped["pipelines"][0].keys()) == ["activities", "name"] + + +def test_motifs_are_decoupled_from_the_lineage_block() -> None: + """Motifs ride as their own pipeline key, never nested under ``lineage``. + + A pipeline that has both derived lineage and a detected motif emits both, and + the lineage block's own (convert-time) motif slot stays empty and separate. + """ + lineage = Lineage( + data_edges=[DataEdge(source_task_key="a", target_task_key="b", match_kind="identity", match_key="cat.sch.tbl")] + ) + graph = SourceGraph(name="g", source="unit", tasks=[_node("a", "Notebook", "deterministic")], lineage=lineage) + motif = _motif("scd_type_2", "dlt_apply_changes", ["a"]) + + inventory = build_source_inventory([graph], source="unit", source_dir="/tmp", motifs_by_pipeline={"g": [motif]}) + entry = inventory["pipelines"][0] + + assert entry["motifs"][0]["motif_id"] == "scd_type_2" + # The lineage block is present but its own motif slot is untouched and empty. + assert entry["lineage"]["motifs"] == [] + + +def test_exact_duplicate_motifs_collapse_but_overlapping_matches_survive() -> None: + """Exact duplicates dedupe to the first; overlapping-but-distinct matches all survive. + + A detector can report the same match twice (same motif over the same members), + which must collapse to one -- but a match that merely *overlaps* (same motif, + a different member set) is a distinct detection and must be kept. Order is + first-seen, and the member comparison is a set (order-insensitive). + """ + graph = SourceGraph( + name="g1", + source="unit", + tasks=[ + _node("copy", "Copy", "deterministic"), + _node("notify_a", "WebActivity", "deterministic"), + _node("notify_b", "WebActivity", "deterministic"), + ], + ) + exact = _motif("activity_and_notify", "task_with_notification", ["copy", "notify_a"]) + exact_dupe = _motif( + "activity_and_notify", + "task_with_notification", + ["notify_a", "copy"], # same member set, different order -> the same motif + notes=["a different note on the same match"], + ) + overlapping = _motif("activity_and_notify", "task_with_notification", ["copy", "notify_b"]) + + inventory = build_source_inventory( + [graph], + source="unit", + source_dir="/tmp", + motifs_by_pipeline={"g1": [exact, exact_dupe, overlapping]}, + ) + motifs = inventory["pipelines"][0]["motifs"] + + # The exact duplicate collapsed; the overlapping-but-distinct match survived, in first-seen order. + assert [frozenset(motif["member_task_keys"]) for motif in motifs] == [ + frozenset({"copy", "notify_a"}), + frozenset({"copy", "notify_b"}), + ] + # First occurrence is the one kept (its empty notes, not the duplicate's note). + assert motifs[0]["confidence_notes"] == [] diff --git a/tests/unit/test_lineage_substrate.py b/tests/unit/test_lineage_substrate.py new file mode 100644 index 00000000..1f1d8a8c --- /dev/null +++ b/tests/unit/test_lineage_substrate.py @@ -0,0 +1,454 @@ +"""Unit tests for the source-neutral lineage substrate (#61). + +Covers the pure derivation (:mod:`flowx.lineage`) -- control fan-out, nested and +Switch recursion, the identity-vs-signature join tiers, no self-edges, no +duplicate edges -- and the ``ir_serde`` round-trip for the new IR types plus the +new ``Activity`` base fields, including a MotifActivity round-trip that proves the +R1 ``motif_id`` collision is handled. +""" + +from __future__ import annotations + +import json + +from flowx.bundler.dab_writer import pipeline_dict_to_ir +from flowx.ir_serde import pipeline_to_dict +from flowx.lineage import ( + build_control_edges, + build_data_edges, + build_lineage, + build_motif_annotations, + with_lineage, +) +from flowx.models.ir import ( + ControlEdge, + DataAsset, + DataEdge, + ExecutePipelineActivity, + ForEachActivity, + IfConditionActivity, + Lineage, + MotifActivity, + MotifAnnotation, + NotebookActivity, + Pipeline, + RunJobActivity, + SwitchActivity, + SwitchCase, + WaitActivity, +) + + +def _notebook(task_key: str, *, reads=None, writes=None, motif_id=None) -> NotebookActivity: + return NotebookActivity( + name=task_key, + task_key=task_key, + notebook_path=f"/Shared/{task_key}", + data_reads=list(reads or []), + data_writes=list(writes or []), + motif_id=motif_id, + ) + + +def _execute(task_key: str, callee: str, *, wait: bool = True) -> ExecutePipelineActivity: + return ExecutePipelineActivity(name=task_key, task_key=task_key, pipeline_name=callee, wait_on_completion=wait) + + +# --------------------------------------------------------------------------- # +# Control-edge derivation +# --------------------------------------------------------------------------- # + + +def test_control_edges_fan_out_and_nested_switch_recursion(): + """ExecutePipeline calls are found at top level and inside ForEach/If/Switch.""" + pipeline = Pipeline( + name="parent", + tasks=[ + _execute("call_a", "child_a"), + ForEachActivity( + name="fe", + task_key="fe", + items_expression="@x", + inner_activities=[_execute("call_b", "child_b")], + ), + IfConditionActivity( + name="cond", + task_key="cond", + op="equals", + left="@a", + right="@b", + if_true_activities=[_execute("call_c", "child_c")], + if_false_activities=[_execute("call_d", "child_d")], + ), + SwitchActivity( + name="sw", + task_key="sw", + on_expression="@e", + cases=[SwitchCase(value="one", activities=[_execute("call_e", "child_e")])], + default_activities=[_execute("call_f", "child_f")], + ), + ], + ) + + edges = build_control_edges(pipeline) + + targets = sorted(edge.target_workflow for edge in edges) + assert targets == ["child_a", "child_b", "child_c", "child_d", "child_e", "child_f"] + assert all(edge.source_workflow == "parent" for edge in edges) + # Each call site keeps its own via_task_key (fan-out preserved). + assert {edge.via_task_key for edge in edges} == { + "call_a", + "call_b", + "call_c", + "call_d", + "call_e", + "call_f", + } + + +def test_control_edges_run_job_activity_is_source_neutral(): + """A RunJobActivity (Airflow) produces a control edge just like ExecutePipeline.""" + pipeline = Pipeline( + name="dag_main", + tasks=[RunJobActivity(name="run", task_key="run", job_name="downstream_job")], + ) + + edges = build_control_edges(pipeline) + + assert len(edges) == 1 + assert edges[0].source_workflow == "dag_main" + assert edges[0].target_workflow == "downstream_job" + assert edges[0].via_task_key == "run" + assert edges[0].wait_for_completion is None + assert edges[0].resolved is True + + +def test_control_edges_unresolved_callee_is_recorded_not_dropped(): + """An empty callee is kept with resolved=False rather than silently dropped.""" + pipeline = Pipeline(name="parent", tasks=[_execute("call", "")]) + + edges = build_control_edges(pipeline) + + assert len(edges) == 1 + assert edges[0].target_workflow == "" + assert edges[0].resolved is False + + +def test_control_edges_no_self_edge(): + """A pipeline invoking itself produces no edge.""" + pipeline = Pipeline(name="loop", tasks=[_execute("call", "loop")]) + + assert build_control_edges(pipeline) == [] + + +def test_control_edges_no_duplicate_from_recursion(): + """A single call site nested in a container is emitted exactly once.""" + pipeline = Pipeline( + name="parent", + tasks=[ + ForEachActivity( + name="fe", + task_key="fe", + items_expression="@x", + inner_activities=[_execute("call", "child")], + ) + ], + ) + + edges = build_control_edges(pipeline) + + assert len(edges) == 1 + assert edges[0].target_workflow == "child" + + +# --------------------------------------------------------------------------- # +# Data-edge derivation: identity vs signature tiers +# --------------------------------------------------------------------------- # + + +def test_data_edges_identity_tier_joins_across_different_signatures(): + """Two assets with the same resolved identity match even when their names differ.""" + pipeline = Pipeline( + name="p", + tasks=[ + _notebook("writer", writes=[DataAsset(signature="ds_out", identity="curated.orders")]), + _notebook("reader", reads=[DataAsset(signature="ds_in_other_name", identity="curated.orders")]), + ], + ) + + edges = build_data_edges(pipeline) + + assert len(edges) == 1 + assert edges[0].source_task_key == "writer" + assert edges[0].target_task_key == "reader" + assert edges[0].match_kind == "identity" + assert edges[0].match_key == "curated.orders" + assert edges[0].identity == "curated.orders" + + +def test_data_edges_signature_tier_when_identity_unresolved(): + """When identity is unresolvable, matching falls back to the neutral signature.""" + pipeline = Pipeline( + name="p", + tasks=[ + _notebook("writer", writes=[DataAsset(signature="shared_ds")]), + _notebook("reader", reads=[DataAsset(signature="shared_ds")]), + ], + ) + + edges = build_data_edges(pipeline) + + assert len(edges) == 1 + assert edges[0].match_kind == "signature" + assert edges[0].match_key == "shared_ds" + assert edges[0].identity is None + + +def test_data_edges_distinct_identities_do_not_fall_back_to_signature(): + """Two resolved-but-different identities never manufacture a signature edge (#36).""" + pipeline = Pipeline( + name="p", + tasks=[ + _notebook("writer", writes=[DataAsset(signature="shared", identity="a.first")]), + _notebook("reader", reads=[DataAsset(signature="shared", identity="b.second")]), + ], + ) + + assert build_data_edges(pipeline) == [] + + +def test_data_edges_fan_out_one_writer_many_readers(): + """One producer handing off to several consumers yields one edge each.""" + pipeline = Pipeline( + name="p", + tasks=[ + _notebook("writer", writes=[DataAsset(signature="ds", identity="x.y")]), + _notebook("reader_one", reads=[DataAsset(signature="ds", identity="x.y")]), + _notebook("reader_two", reads=[DataAsset(signature="ds", identity="x.y")]), + ], + ) + + edges = build_data_edges(pipeline) + + assert sorted(edge.target_task_key for edge in edges) == ["reader_one", "reader_two"] + assert all(edge.source_task_key == "writer" for edge in edges) + + +def test_data_edges_no_self_edge(): + """An activity that both writes and reads the same asset does not edge to itself.""" + pipeline = Pipeline( + name="p", + tasks=[ + _notebook( + "roundtrip", + writes=[DataAsset(signature="ds", identity="x.y")], + reads=[DataAsset(signature="ds", identity="x.y")], + ) + ], + ) + + assert build_data_edges(pipeline) == [] + + +def test_data_edges_no_duplicate_from_repeated_asset(): + """A producer listing the same asset twice still yields a single edge.""" + pipeline = Pipeline( + name="p", + tasks=[ + _notebook( + "writer", + writes=[DataAsset(signature="ds", identity="x.y"), DataAsset(signature="ds", identity="x.y")], + ), + _notebook("reader", reads=[DataAsset(signature="ds", identity="x.y")]), + ], + ) + + edges = build_data_edges(pipeline) + + assert len(edges) == 1 + + +def test_data_edges_nested_switch_recursion(): + """A producer buried in a Switch case hands off to a top-level consumer.""" + pipeline = Pipeline( + name="p", + tasks=[ + SwitchActivity( + name="sw", + task_key="sw", + on_expression="@e", + cases=[ + SwitchCase( + value="one", + activities=[_notebook("writer", writes=[DataAsset(signature="ds", identity="x.y")])], + ) + ], + default_activities=[], + ), + _notebook("reader", reads=[DataAsset(signature="ds", identity="x.y")]), + ], + ) + + edges = build_data_edges(pipeline) + + assert len(edges) == 1 + assert edges[0].source_task_key == "writer" + assert edges[0].target_task_key == "reader" + + +# --------------------------------------------------------------------------- # +# Motif annotations + composition + purity +# --------------------------------------------------------------------------- # + + +def test_build_motif_annotations_groups_members_by_tag(): + """A MotifActivity plus tagged members become one annotation over their task keys.""" + pipeline = Pipeline( + name="p", + tasks=[ + MotifActivity( + name="motif", + task_key="motif_auto_loader", + motif_id="auto_loader", + display_name="Auto Loader", + databricks_replacement="auto_loader", + matched_activity_names=["Copy A", "Copy B"], + confidence_notes=["matched on file source"], + ), + _notebook("member", motif_id="auto_loader"), + _notebook("unrelated"), + ], + ) + + annotations = build_motif_annotations(pipeline) + + assert len(annotations) == 1 + assert annotations[0].motif_id == "auto_loader" + assert annotations[0].member_task_keys == ["motif_auto_loader", "member"] + assert annotations[0].display_name == "Auto Loader" + assert annotations[0].databricks_replacement == "auto_loader" + + +def test_with_lineage_is_pure(): + """with_lineage returns a new pipeline and never mutates the input.""" + pipeline = Pipeline(name="p", tasks=[_notebook("n")]) + lineage = build_lineage(pipeline) + + updated = with_lineage(pipeline, lineage) + + assert pipeline.lineage is None + assert updated is not pipeline + assert updated.lineage is lineage + + +# --------------------------------------------------------------------------- # +# ir_serde round-trips +# --------------------------------------------------------------------------- # + + +def test_serde_round_trip_new_activity_fields_and_lineage_block(): + """data_reads/data_writes/motif_id and the lineage block survive JSON round-trip.""" + pipeline = Pipeline( + name="p", + tasks=[ + _notebook( + "writer", + writes=[DataAsset(signature="ds_out", identity="curated.orders", asset_type="table")], + motif_id="auto_loader", + ), + _notebook( + "reader", + reads=[ + DataAsset( + signature="ds_in", + identity="curated.orders", + asset_type="table", + properties={"format": "delta"}, + ) + ], + ), + WaitActivity(name="pause", task_key="pause", wait_time_seconds=5), + ], + lineage=Lineage( + control_edges=[ + ControlEdge( + source_workflow="p", + target_workflow="child", + via_task_key="writer", + wait_for_completion=True, + resolved=True, + ) + ], + data_edges=[ + DataEdge( + source_task_key="writer", + target_task_key="reader", + match_kind="identity", + match_key="curated.orders", + identity="curated.orders", + asset_type="table", + ) + ], + motifs=[ + MotifAnnotation( + motif_id="auto_loader", + member_task_keys=["writer"], + display_name="Auto Loader", + databricks_replacement="auto_loader", + notes=["note"], + ) + ], + ), + ) + + reloaded, _ = pipeline_dict_to_ir(json.loads(json.dumps(pipeline_to_dict(pipeline)))) + + writer = reloaded.tasks[0] + reader = reloaded.tasks[1] + assert writer.motif_id == "auto_loader" + assert writer.data_writes == [DataAsset(signature="ds_out", identity="curated.orders", asset_type="table")] + assert reader.data_reads == [ + DataAsset(signature="ds_in", identity="curated.orders", asset_type="table", properties={"format": "delta"}) + ] + # A task without lineage fields rehydrates to empty lists / None, never missing. + assert reloaded.tasks[2].data_reads == [] + assert reloaded.tasks[2].data_writes == [] + assert reloaded.tasks[2].motif_id is None + + assert reloaded.lineage == pipeline.lineage + + +def test_serde_round_trip_motif_activity_no_kwarg_collision_r1(): + """A MotifActivity round-trips without the R1 'multiple values for motif_id' TypeError.""" + pipeline = Pipeline( + name="p", + tasks=[ + MotifActivity( + name="motif", + task_key="motif_auto_loader", + motif_id="auto_loader", + display_name="Auto Loader", + databricks_replacement="auto_loader", + matched_activity_names=["Copy A", "Copy B"], + data_reads=[DataAsset(signature="src", identity="raw.src")], + ) + ], + ) + + reloaded, _ = pipeline_dict_to_ir(json.loads(json.dumps(pipeline_to_dict(pipeline)))) + + task = reloaded.tasks[0] + assert isinstance(task, MotifActivity) + assert task.motif_id == "auto_loader" + assert task.display_name == "Auto Loader" + assert task.matched_activity_names == ["Copy A", "Copy B"] + assert task.data_reads == [DataAsset(signature="src", identity="raw.src")] + + +def test_lineage_block_always_emits_lists_never_null(): + """An attached empty lineage serialises its edge collections as lists, not null.""" + pipeline = with_lineage(Pipeline(name="p", tasks=[_notebook("n")]), Lineage()) + + serialised = pipeline_to_dict(pipeline) + + assert serialised["lineage"] == {"control_edges": [], "data_edges": [], "motifs": []} From a8aa8e977766151bd69dcd069b3cc96f1e13f870 Mon Sep 17 00:00:00 2001 From: Matthew Moorcroft Date: Fri, 18 Sep 2026 11:30:18 +0100 Subject: [PATCH 02/25] Discover: loud 0-pipeline warning + self-describing activity-count units FIX 1: When ADF discover parses zero pipelines (e.g. --source-dir pointed at a directory of ARM templates with no recognized pipelines/ layout), emit a loud stderr warning instead of finishing with a success-shaped "Loaded 0 pipeline(s)". The warning steers the operator to pass the ARM template as the file path or supply the expected export layout. Exit code for the 0-pipeline case is unchanged (still 0); behaviour for valid dirs/files is untouched. FIX 7: Label the discover summary's "Total activities" count as raw activities including nested, and note that convert reports top-level task-units after motif collapse, so the smaller convert number is not mistaken for a discrepancy. Console-output only -- inventory.json keys are byte-compatible (unchanged). Co-authored-by: Isaac --- src/flowx/sources/adf/loader.py | 46 ++++++++++++++++++++++++++++++++- tests/unit/test_adf_loader.py | 30 +++++++++++++++++++++ 2 files changed, 75 insertions(+), 1 deletion(-) diff --git a/src/flowx/sources/adf/loader.py b/src/flowx/sources/adf/loader.py index 27ce7c1b..c909da3f 100644 --- a/src/flowx/sources/adf/loader.py +++ b/src/flowx/sources/adf/loader.py @@ -8,6 +8,7 @@ import logging import re import shutil +import sys from pathlib import Path from typing import Any @@ -1027,6 +1028,38 @@ def clear_stale_outputs(output_dir: Path) -> None: (output_dir / filename).unlink(missing_ok=True) +def _warn_no_pipelines_found(source_dir: Path) -> None: + """Prints a loud stderr warning when discover parses zero pipelines. + + The usual cause is pointing ``--source-dir`` at a directory that holds ARM-template + ``.json`` file(s) instead of the expected ADF export layout (a ``pipelines/`` folder), or + passing a directory when a single ARM template file was meant. We tailor the guidance to + whichever case we can detect, so an empty run never looks like a successful one. + """ + source_path = Path(source_dir) + top_level_json = sorted(source_path.glob("*.json")) if source_path.is_dir() else [] + + banner = "!" * 72 + lines = [banner, "WARNING: discover loaded 0 pipelines -- nothing was migrated."] + if top_level_json: + lines.append( + f"Found {len(top_level_json)} top-level .json file(s) in {source_path} but no " + "recognized ADF export layout (no 'pipelines/' directory)." + ) + lines.append( + "If these are ARM templates, pass the ARM template file directly as the " + "--source-dir path (a single .json file), not the containing directory." + ) + else: + lines.append( + f"No ADF pipelines were found under {source_path}. Pass an ARM template file as " + "the path, or point --source-dir at a directory with the expected ADF export " + "layout ('pipelines/', 'datasets/', 'linked_services/', ...)." + ) + lines.append(banner) + print("\n".join(lines), file=sys.stderr) + + def main(argv: list[str] | None = None) -> int: """Discover-phase entry point: load ADF, build the inventory + profile report. @@ -1059,6 +1092,13 @@ def main(argv: list[str] | None = None) -> int: definitions = load_adf_definitions(args.source_dir) logger.info("Loaded %d pipeline(s) from %s", len(definitions.pipelines), args.source_dir) + # A directory of ARM templates with no recognized ``pipelines/`` layout parses to zero + # pipelines and would otherwise finish with a success-shaped "Loaded 0 pipeline(s)". Fail + # loud (stderr) so the operator notices the export layout / path is wrong rather than + # trusting an empty-but-green run. + if not definitions.pipelines: + _warn_no_pipelines_found(args.source_dir) + # Filter to a single pipeline when --pipeline is specified if args.pipeline: matched = [pipeline for pipeline in definitions.pipelines if pipeline.name == args.pipeline] @@ -1120,7 +1160,11 @@ def main(argv: list[str] | None = None) -> int: print("\nADF Profile Summary") print("===================") print(f"Pipelines parsed: {summary['pipeline_count']}") - print(f"Total activities: {summary['activity_count']}") + print(f"Total activities: {summary['activity_count']} (raw activities, incl. nested)") + # Discover counts every activity, including those nested inside ForEach/If/Switch. The + # convert phase reports top-level task-units *after* motif collapse, so its count is + # smaller -- label the units here so the two numbers are not mistaken for a discrepancy. + print(" (convert reports top-level task-units after motif collapse; expect fewer there)") print("\nStrategy Breakdown:") print(f" Deterministic: {summary['deterministic_count']}") print(f" Agentic: {summary['agentic_count']}") diff --git a/tests/unit/test_adf_loader.py b/tests/unit/test_adf_loader.py index e6730515..9f94d29c 100644 --- a/tests/unit/test_adf_loader.py +++ b/tests/unit/test_adf_loader.py @@ -20,6 +20,7 @@ classify_activity, clear_stale_outputs, load_adf_definitions, + main, ) # --------------------------------------------------------------------------- @@ -554,3 +555,32 @@ def test_idempotent_on_empty_dir(self, tmp_path): """Clearing a directory with no flowx artifacts is a no-op (no error).""" clear_stale_outputs(tmp_path) assert list(tmp_path.iterdir()) == [] + + +class TestZeroPipelineWarning: + """discover must fail loud (stderr) when it parses zero pipelines.""" + + def test_arm_template_directory_warns_loudly(self, tmp_path, capsys): + """A dir holding ARM-template .json (no pipelines/ layout) loads 0 pipelines + warns.""" + source_dir = tmp_path / "arm_export" + source_dir.mkdir() + # An ARM template dropped at the top level, not the expected pipelines/ layout. + (source_dir / "ARMTemplateForFactory.json").write_text(json.dumps({"resources": []}), encoding="utf-8") + + exit_code = main(["--source-dir", str(source_dir), "--output-dir", str(tmp_path / "out")]) + + # Behaviour for the 0-pipeline case is unchanged (still exits 0); the deliverable is the + # loud stderr warning that steers the operator to pass the ARM template as the file path. + assert exit_code == 0 + stderr = capsys.readouterr().err + assert "0 pipelines" in stderr + assert "ARM template" in stderr + assert ".json" in stderr + + def test_valid_directory_does_not_warn(self, fixtures_dir, tmp_path, capsys): + """A valid ADF export loads pipelines and emits no zero-pipeline warning.""" + exit_code = main(["--source-dir", str(fixtures_dir), "--output-dir", str(tmp_path / "out")]) + + assert exit_code == 0 + stderr = capsys.readouterr().err + assert "discover loaded 0 pipelines" not in stderr From 79c41dd8fb342f860d805163ece7557fcec2262e Mon Sep 17 00:00:00 2001 From: Matthew Moorcroft Date: Fri, 18 Sep 2026 12:43:58 +0100 Subject: [PATCH 03/25] Discover 0-pipeline warning: name the user-facing --adf-source-path flag Cross-review (non-blocking): the loud 0-pipeline warning referenced the loader's internal --source-dir flag, but that is not what a user passes -- the discover runner takes --adf-source-path (the generic --source-path alias normalises to --source-dir internally). Name --adf-source-path in the guidance so the operator is pointed at the flag they actually use. Test asserts the user-facing flag is named and the internal one is not. Co-authored-by: Isaac --- src/flowx/sources/adf/loader.py | 12 +++++++----- tests/unit/test_adf_loader.py | 3 +++ 2 files changed, 10 insertions(+), 5 deletions(-) diff --git a/src/flowx/sources/adf/loader.py b/src/flowx/sources/adf/loader.py index c909da3f..51f582a2 100644 --- a/src/flowx/sources/adf/loader.py +++ b/src/flowx/sources/adf/loader.py @@ -1031,10 +1031,12 @@ def clear_stale_outputs(output_dir: Path) -> None: def _warn_no_pipelines_found(source_dir: Path) -> None: """Prints a loud stderr warning when discover parses zero pipelines. - The usual cause is pointing ``--source-dir`` at a directory that holds ARM-template + The usual cause is pointing ``--adf-source-path`` at a directory that holds ARM-template ``.json`` file(s) instead of the expected ADF export layout (a ``pipelines/`` folder), or passing a directory when a single ARM template file was meant. We tailor the guidance to - whichever case we can detect, so an empty run never looks like a successful one. + whichever case we can detect, so an empty run never looks like a successful one. The message + names ``--adf-source-path`` -- the user-facing flag the discover runner takes -- not the + loader's internal ``--source-dir`` it normalises to. """ source_path = Path(source_dir) top_level_json = sorted(source_path.glob("*.json")) if source_path.is_dir() else [] @@ -1048,13 +1050,13 @@ def _warn_no_pipelines_found(source_dir: Path) -> None: ) lines.append( "If these are ARM templates, pass the ARM template file directly as the " - "--source-dir path (a single .json file), not the containing directory." + "--adf-source-path (a single .json file), not the containing directory." ) else: lines.append( f"No ADF pipelines were found under {source_path}. Pass an ARM template file as " - "the path, or point --source-dir at a directory with the expected ADF export " - "layout ('pipelines/', 'datasets/', 'linked_services/', ...)." + "the --adf-source-path, or point --adf-source-path at a directory with the expected " + "ADF export layout ('pipelines/', 'datasets/', 'linked_services/', ...)." ) lines.append(banner) print("\n".join(lines), file=sys.stderr) diff --git a/tests/unit/test_adf_loader.py b/tests/unit/test_adf_loader.py index 9f94d29c..248f3e81 100644 --- a/tests/unit/test_adf_loader.py +++ b/tests/unit/test_adf_loader.py @@ -576,6 +576,9 @@ def test_arm_template_directory_warns_loudly(self, tmp_path, capsys): assert "0 pipelines" in stderr assert "ARM template" in stderr assert ".json" in stderr + # The guidance names the user-facing flag (--adf-source-path), not the loader-internal one. + assert "--adf-source-path" in stderr + assert "--source-dir" not in stderr def test_valid_directory_does_not_warn(self, fixtures_dir, tmp_path, capsys): """A valid ADF export loads pipelines and emits no zero-pipeline warning.""" From e39a05fd274fe64939a7e2617d1395fc5d1bcf8c Mon Sep 17 00:00:00 2001 From: Matthew Moorcroft Date: Mon, 5 Oct 2026 12:10:49 +0100 Subject: [PATCH 04/25] Persist ADF source graphs at discover through the shared envelope ADF discover built SourceGraphs only in memory and wrote the lossy inventory projection. It now also writes metadata/source_graphs.json with write_source_graphs (contract_version "1", per-graph and document hashes), the same file Airflow discover writes, so a later phase can read exactly what discover saw without parsing the source again. Convert does not read it yet. Co-authored-by: Isaac --- src/flowx/sources/adf/loader.py | 11 +++++++++-- tests/unit/test_adf_inventory_superset.py | 18 +++++++++++++++++- 2 files changed, 26 insertions(+), 3 deletions(-) diff --git a/src/flowx/sources/adf/loader.py b/src/flowx/sources/adf/loader.py index 51f582a2..b20e8fbd 100644 --- a/src/flowx/sources/adf/loader.py +++ b/src/flowx/sources/adf/loader.py @@ -1078,7 +1078,7 @@ def main(argv: list[str] | None = None) -> int: default=Path("./flowx_output"), help=( "Migration output directory. Profile artifacts are written into its " - "metadata/ subfolder (inventory.json, profile_report.csv, .arm.json)." + "metadata/ subfolder (inventory.json, source_graphs.json, profile_report.csv, .arm.json)." ), ) parser.add_argument( @@ -1121,10 +1121,11 @@ def main(argv: list[str] | None = None) -> int: ) logger.info("Filtered to pipeline: %s", args.pipeline) - # Inventory JSON is projected from the shared discovery AST via the + # Inventory JSON is projected from the discovery graph contract via the # source-agnostic emitter (so ADF and Airflow emit one shape); imported here # to avoid a module-level cycle (discovery_mapping imports this loader). from flowx.discovery_inventory import build_source_inventory + from flowx.discovery_serde import SOURCE_GRAPHS_FILENAME, write_source_graphs from flowx.models.discovery import SOURCE_ADF from flowx.sources.adf.discovery_mapping import adf_definitions_to_source_graphs @@ -1150,6 +1151,12 @@ def main(argv: list[str] | None = None) -> int: inventory_path.write_text(json.dumps(inventory_dict, indent=2), encoding="utf-8") logger.info("Wrote inventory to %s", inventory_path) + # Persist the full graphs too, not just the lossy inventory projection, so a later phase can + # read exactly what discover saw without parsing the source again. + source_graphs_path = metadata_dir / SOURCE_GRAPHS_FILENAME + write_source_graphs(source_graphs_path, source_graphs, source=SOURCE_ADF) + logger.info("Wrote source graphs to %s", source_graphs_path) + profile_rows = build_profile_rows(definitions, motifs_by_pipeline) csv_path = metadata_dir / "profile_report.csv" write_profile_csv(profile_rows, csv_path) diff --git a/tests/unit/test_adf_inventory_superset.py b/tests/unit/test_adf_inventory_superset.py index 8963cece..6bd7eac2 100644 --- a/tests/unit/test_adf_inventory_superset.py +++ b/tests/unit/test_adf_inventory_superset.py @@ -1,4 +1,4 @@ -"""Consumer-safety tests for the ADF inventory emitted via the shared AST + emitter. +"""Consumer-safety tests for the ADF inventory emitted via the discovery graph + emitter. Two guarantees: @@ -17,6 +17,7 @@ import json from pathlib import Path +from flowx.discovery_serde import read_source_graphs from flowx.reporting.coverage import build_coverage_rows from flowx.sources.adf.loader import build_inventory, load_adf_definitions, main @@ -191,3 +192,18 @@ def test_coverage_output_matches_golden(tmp_path: Path) -> None: rows = json.loads(json.dumps(build_coverage_rows(metadata), sort_keys=True)) golden = json.loads(GOLDEN_COVERAGE.read_text()) assert rows == golden + + +def test_discover_persists_source_graphs_that_match_the_inventory(tmp_path: Path) -> None: + """Discover writes the full, hashed source graphs beside the inventory, one per parsed pipeline.""" + metadata = _run_discover(tmp_path) + + document = json.loads((metadata / "source_graphs.json").read_text(encoding="utf-8")) + graphs = read_source_graphs(metadata / "source_graphs.json") + inventory = json.loads((metadata / "inventory.json").read_text(encoding="utf-8")) + + assert document["contract_version"] == "1" + assert document["source"] == "adf" + assert len(document["graph_sha256"]) == len(graphs) == inventory["summary"]["pipeline_count"] + listed = {pipeline["name"] for pipeline in inventory["pipelines"]} + assert listed <= {graph.name for graph in graphs} From db4641964293ebed457b1fd9f3614f93117f4e2f Mon Sep 17 00:00:00 2001 From: Matthew Moorcroft Date: Mon, 5 Oct 2026 12:10:51 +0100 Subject: [PATCH 05/25] Call the ADF mapper's target the discovery graph contract Docstring and comment wording only. Co-authored-by: Isaac --- src/flowx/sources/adf/discovery_mapping.py | 2 +- tests/unit/test_adf_discovery_mapping.py | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/src/flowx/sources/adf/discovery_mapping.py b/src/flowx/sources/adf/discovery_mapping.py index 0555a312..d96fb869 100644 --- a/src/flowx/sources/adf/discovery_mapping.py +++ b/src/flowx/sources/adf/discovery_mapping.py @@ -1,4 +1,4 @@ -"""Map the ADF AST onto the shared, source-faithful discovery AST. +"""Map the ADF AST onto the source-neutral discovery graph contract (``SourceGraph``). This is the ADF half of the discovery contract (issue #62): it turns the typed ADF AST (:mod:`flowx.models.adf_ast`) into the shared diff --git a/tests/unit/test_adf_discovery_mapping.py b/tests/unit/test_adf_discovery_mapping.py index 71d72b3b..a8037f81 100644 --- a/tests/unit/test_adf_discovery_mapping.py +++ b/tests/unit/test_adf_discovery_mapping.py @@ -1,4 +1,4 @@ -"""Tests for the ADF -> shared discovery AST mapper (:mod:`flowx.sources.adf.discovery_mapping`). +"""Tests for the ADF -> discovery graph mapper (:mod:`flowx.sources.adf.discovery_mapping`). Proves the mapping is 1:1 and lossless: every activity becomes one discovery node, the ADF type is retained verbatim as ``native_type`` / ``original_type``, From b9e330591f4ee374c9b280affb3407cfbf7b1622 Mon Sep 17 00:00:00 2001 From: Matthew Moorcroft Date: Mon, 5 Oct 2026 16:57:08 +0100 Subject: [PATCH 06/25] Save ADF motifs in the source graphs and project the inventory from the saved file ADF discover now records each pipeline's detected motifs on its source graph before writing source_graphs.json, so they are saved and hashed with the graph, then projects inventory.json from those graphs and records the file's document hash as source_graphs_sha256. Detection still runs once on the ADF definitions and the profile report is unchanged. Co-authored-by: Isaac --- src/flowx/sources/adf/loader.py | 49 ++++++++++++++++++----- tests/unit/test_adf_inventory_superset.py | 18 +++++++++ 2 files changed, 58 insertions(+), 9 deletions(-) diff --git a/src/flowx/sources/adf/loader.py b/src/flowx/sources/adf/loader.py index b20e8fbd..f1dff245 100644 --- a/src/flowx/sources/adf/loader.py +++ b/src/flowx/sources/adf/loader.py @@ -9,6 +9,7 @@ import re import shutil import sys +from collections.abc import Mapping from pathlib import Path from typing import Any @@ -29,6 +30,8 @@ InventoryItem, TranslationStrategy, ) +from flowx.models.discovery import SourceGraph +from flowx.models.ir import Lineage, MotifAnnotation from flowx.models.motifs import DetectedMotif logger = logging.getLogger(__name__) @@ -904,6 +907,32 @@ def detect_motifs_by_pipeline(definitions: AdfDefinitions) -> dict[str, list[Det return results +def attach_motifs_to_graphs(graphs: list[SourceGraph], motifs_by_pipeline: Mapping[str, list[DetectedMotif]]) -> None: + """Record each pipeline's detected motifs on its source graph as lineage motif annotations. + + Motifs are deterministic facts about the pipeline, so they belong in the saved + graph next to its edges. Detection itself still runs on the ADF definitions; + this only copies the results onto the graph, member activities untouched. + """ + for graph in graphs: + detections = motifs_by_pipeline.get(graph.name) + if not detections: + continue + if graph.lineage is None: + graph.lineage = Lineage() + graph.lineage.motifs = [ + MotifAnnotation( + motif_id=motif.definition.motif_id, + member_task_keys=list(motif.matched_activities), + display_name=motif.definition.display_name, + databricks_replacement=motif.definition.databricks_replacement, + notes=list(motif.confidence_notes), + source_type_hint=motif.source_type_hint, + ) + for motif in detections + ] + + def build_profile_rows( definitions: AdfDefinitions, motifs_by_pipeline: dict[str, list[DetectedMotif]] | None = None, @@ -1136,9 +1165,17 @@ def main(argv: list[str] | None = None) -> int: inventory_path = metadata_dir / "inventory.json" source_graphs = adf_definitions_to_source_graphs(definitions) - # Detect motifs once and share the result: the inventory surfaces the full - # detections additively, the profile report counts them -- from one pass. + # Detect motifs once and share the result: the graphs carry them (so they are saved and + # hashed with each graph), the inventory surfaces them, the profile report counts them. motifs_by_pipeline = detect_motifs_by_pipeline(definitions) + attach_motifs_to_graphs(source_graphs, motifs_by_pipeline) + + # Persist the full graphs first, so the inventory projected from them can record which saved + # graph it describes, and a later phase can read exactly what discover saw. + source_graphs_path = metadata_dir / SOURCE_GRAPHS_FILENAME + source_graphs_document = write_source_graphs(source_graphs_path, source_graphs, source=SOURCE_ADF) + logger.info("Wrote source graphs to %s", source_graphs_path) + inventory_dict = build_source_inventory( source_graphs, source=SOURCE_ADF, @@ -1146,17 +1183,11 @@ def main(argv: list[str] | None = None) -> int: # ADF has historically omitted zero-activity pipelines from the per-pipeline # listing while still counting them in summary.pipeline_count; preserve that. include_empty_pipelines=False, - motifs_by_pipeline=motifs_by_pipeline, + source_graphs_sha256=source_graphs_document["document_sha256"], ) inventory_path.write_text(json.dumps(inventory_dict, indent=2), encoding="utf-8") logger.info("Wrote inventory to %s", inventory_path) - # Persist the full graphs too, not just the lossy inventory projection, so a later phase can - # read exactly what discover saw without parsing the source again. - source_graphs_path = metadata_dir / SOURCE_GRAPHS_FILENAME - write_source_graphs(source_graphs_path, source_graphs, source=SOURCE_ADF) - logger.info("Wrote source graphs to %s", source_graphs_path) - profile_rows = build_profile_rows(definitions, motifs_by_pipeline) csv_path = metadata_dir / "profile_report.csv" write_profile_csv(profile_rows, csv_path) diff --git a/tests/unit/test_adf_inventory_superset.py b/tests/unit/test_adf_inventory_superset.py index 6bd7eac2..b82435c5 100644 --- a/tests/unit/test_adf_inventory_superset.py +++ b/tests/unit/test_adf_inventory_superset.py @@ -207,3 +207,21 @@ def test_discover_persists_source_graphs_that_match_the_inventory(tmp_path: Path assert len(document["graph_sha256"]) == len(graphs) == inventory["summary"]["pipeline_count"] listed = {pipeline["name"] for pipeline in inventory["pipelines"]} assert listed <= {graph.name for graph in graphs} + assert inventory["source_graphs_sha256"] == document["document_sha256"] + + +def test_detected_motifs_are_saved_inside_the_hashed_graphs(tmp_path: Path) -> None: + """Every motif the inventory lists is also recorded on that pipeline's saved graph.""" + metadata = _run_discover(tmp_path) + + graphs = {graph.name: graph for graph in read_source_graphs(metadata / "source_graphs.json")} + inventory = json.loads((metadata / "inventory.json").read_text(encoding="utf-8")) + + with_motifs = [pipeline for pipeline in inventory["pipelines"] if pipeline.get("motifs")] + assert with_motifs, "the ADF fixtures are expected to contain at least one detected motif" + for pipeline in with_motifs: + lineage = graphs[pipeline["name"]].lineage + assert lineage is not None + saved = sorted((motif.motif_id, tuple(motif.member_task_keys)) for motif in lineage.motifs) + listed = sorted((motif["motif_id"], tuple(motif["member_task_keys"])) for motif in pipeline["motifs"]) + assert set(listed) <= set(saved) From b21c15d74dcbdd56e7ab2ec65216c4d62e91d148 Mon Sep 17 00:00:00 2001 From: Matthew Moorcroft Date: Tue, 6 Oct 2026 16:08:26 +0100 Subject: [PATCH 07/25] no-mistakes(review): Qualify ADF dataset identities and share the lineage parser --- src/flowx/bundler/dab_writer.py | 67 +-------- src/flowx/discovery_serde.py | 52 +------ src/flowx/ir_serde.py | 44 ++++++ src/flowx/sources/adf/dataset_lineage.py | 111 ++++++++++++--- tests/unit/test_adf_dataset_lineage.py | 170 +++++++++++++++++++++-- tests/unit/test_lineage_substrate.py | 1 + 6 files changed, 301 insertions(+), 144 deletions(-) diff --git a/src/flowx/bundler/dab_writer.py b/src/flowx/bundler/dab_writer.py index 8cdf20b2..c72fe5ad 100644 --- a/src/flowx/bundler/dab_writer.py +++ b/src/flowx/bundler/dab_writer.py @@ -26,14 +26,12 @@ from flowx.bundler.notebook_writer import write_notebooks from flowx.bundler.prereqs_writer import ManualParameter, build_prereqs, render_setup_md from flowx.bundler.setup_generator import generate_setup_tasks +from flowx.ir_serde import data_asset_from_dict, lineage_from_dict from flowx.models.dab import DabNotebook from flowx.models.ir import ( Activity, AppendVariableActivity, - ControlEdge, CopyActivity, - DataAsset, - DataEdge, DbtFactoryActivity, DeleteActivity, Dependency, @@ -41,10 +39,8 @@ FilterActivity, ForEachActivity, IfConditionActivity, - Lineage, LookupActivity, MotifActivity, - MotifAnnotation, NotebookActivity, Pipeline, PlaceholderActivity, @@ -2026,6 +2022,7 @@ def pipeline_dict_to_ir(pipeline_dict: dict[str, Any]) -> tuple[Pipeline, list[d ): raise ValueError(f"Invalid Pipeline email notification entry: {event!r}") email_notifications[str(event)] = list(recipients) + raw_lineage = pipeline_dict.get("lineage") pipeline = Pipeline( name=pipeline_dict.get("name", "unknown"), @@ -2042,7 +2039,7 @@ def pipeline_dict_to_ir(pipeline_dict: dict[str, Any]) -> tuple[Pipeline, list[d reconciliation_status=pipeline_dict.get("reconciliation_status"), migration_status=pipeline_dict.get("migration_status", "included"), audit=dict(pipeline_dict.get("audit") or {}), - lineage=_reconstruct_lineage(pipeline_dict.get("lineage")), + lineage=lineage_from_dict(raw_lineage) if raw_lineage else None, ) return pipeline, parameters @@ -2319,65 +2316,11 @@ def _common_activity_kwargs(task_ir: dict[str, Any]) -> dict[str, Any]: "compute_mode": task_ir.get("compute_mode"), "notifications": task_ir.get("notifications"), "motif_id": task_ir.get("motif_id"), - "data_reads": _reconstruct_data_assets(task_ir.get("data_reads")), - "data_writes": _reconstruct_data_assets(task_ir.get("data_writes")), + "data_reads": [data_asset_from_dict(asset) for asset in task_ir.get("data_reads") or []], + "data_writes": [data_asset_from_dict(asset) for asset in task_ir.get("data_writes") or []], } -def _reconstruct_data_assets(raw: list[dict[str, Any]] | None) -> list[DataAsset]: - """Rehydrates serialised DataAsset dicts into typed :class:`DataAsset` nodes.""" - if not raw: - return [] - return [ - DataAsset( - signature=asset.get("signature", ""), - identity=asset.get("identity"), - asset_type=asset.get("asset_type"), - properties=dict(asset.get("properties") or {}), - ) - for asset in raw - ] - - -def _reconstruct_lineage(raw: dict[str, Any] | None) -> Lineage | None: - """Rehydrates a serialised lineage block into a typed :class:`Lineage`, or ``None``.""" - if not raw: - return None - return Lineage( - control_edges=[ - ControlEdge( - source_workflow=edge.get("source_workflow", ""), - target_workflow=edge.get("target_workflow", ""), - via_task_key=edge.get("via_task_key", ""), - wait_for_completion=edge.get("wait_for_completion"), - resolved=bool(edge.get("resolved", True)), - ) - for edge in raw.get("control_edges") or [] - ], - data_edges=[ - DataEdge( - source_task_key=edge.get("source_task_key", ""), - target_task_key=edge.get("target_task_key", ""), - match_kind=edge.get("match_kind", ""), - match_key=edge.get("match_key", ""), - identity=edge.get("identity"), - asset_type=edge.get("asset_type"), - ) - for edge in raw.get("data_edges") or [] - ], - motifs=[ - MotifAnnotation( - motif_id=motif.get("motif_id", ""), - member_task_keys=list(motif.get("member_task_keys") or []), - display_name=motif.get("display_name"), - databricks_replacement=motif.get("databricks_replacement"), - notes=list(motif.get("notes") or []), - ) - for motif in raw.get("motifs") or [] - ], - ) - - def _reconstruct_dependencies(raw: list[dict[str, Any]] | None) -> list[Dependency] | None: if not raw: return None diff --git a/src/flowx/discovery_serde.py b/src/flowx/discovery_serde.py index b21c0c64..d0a4051a 100644 --- a/src/flowx/discovery_serde.py +++ b/src/flowx/discovery_serde.py @@ -16,9 +16,8 @@ resolvable physical ``identity`` when there is one (else ``None``), an always-present ``signature``, and an open ``asset_type`` that also covers non-physical / logical / value hand-offs (e.g. an Airflow XCom). A graph's derived -:class:`~flowx.models.ir.Lineage` block is serialised through ``ir_serde``'s -``lineage_to_dict`` for the same reason; its inverse (:func:`_lineage_from_dict`) -lives here because ``ir_serde`` ships only the forward direction. +:class:`~flowx.models.ir.Lineage` block goes through ``ir_serde``'s +``lineage_to_dict`` / ``lineage_from_dict`` pair for the same reason. Every node dict carries a ``node_type`` discriminator (the dataclass name) so a :class:`~flowx.models.discovery.ContainerNode` or @@ -33,7 +32,7 @@ from pathlib import Path from typing import Any -from flowx.ir_serde import data_asset_from_dict, data_asset_to_dict, lineage_to_dict +from flowx.ir_serde import data_asset_from_dict, data_asset_to_dict, lineage_from_dict, lineage_to_dict from flowx.models.discovery import ( ContainerNode, GapNode, @@ -44,7 +43,6 @@ SourceGraph, SourceNode, ) -from flowx.models.ir import ControlEdge, DataEdge, Lineage, MotifAnnotation SOURCE_GRAPHS_FILENAME = "source_graphs.json" SOURCE_GRAPHS_CONTRACT_VERSION = "1" @@ -97,7 +95,7 @@ def source_graph_from_dict(raw: dict[str, Any]) -> SourceGraph: run_timeout_seconds=raw.get("run_timeout_seconds"), tags=list(raw.get("tags") or []), tasks=[_node_from_dict(node) for node in raw.get("tasks") or []], - lineage=_lineage_from_dict(lineage) if lineage else None, + lineage=lineage_from_dict(lineage) if lineage else None, properties=dict(raw.get("properties") or {}), extensions=dict(raw.get("extensions") or {}), raw=raw.get("raw"), @@ -188,48 +186,6 @@ def _canonical_sha256(value: Any) -> str: return hashlib.sha256(canonical.encode("utf-8")).hexdigest() -def _lineage_from_dict(raw: dict[str, Any]) -> Lineage: - """Rehydrate a :class:`Lineage` block from the dict ``ir_serde.lineage_to_dict`` emits. - - The inverse of that forward serialiser (which ``ir_serde`` does not itself - ship), so a discovery graph's lineage round-trips through this module. - """ - return Lineage( - control_edges=[ - ControlEdge( - source_workflow=edge.get("source_workflow", ""), - target_workflow=edge.get("target_workflow", ""), - via_task_key=edge.get("via_task_key", ""), - wait_for_completion=edge.get("wait_for_completion"), - resolved=bool(edge.get("resolved", True)), - ) - for edge in raw.get("control_edges") or [] - ], - data_edges=[ - DataEdge( - source_task_key=edge.get("source_task_key", ""), - target_task_key=edge.get("target_task_key", ""), - match_kind=edge.get("match_kind", ""), - match_key=edge.get("match_key", ""), - identity=edge.get("identity"), - asset_type=edge.get("asset_type"), - ) - for edge in raw.get("data_edges") or [] - ], - motifs=[ - MotifAnnotation( - motif_id=motif.get("motif_id", ""), - member_task_keys=list(motif.get("member_task_keys") or []), - display_name=motif.get("display_name"), - databricks_replacement=motif.get("databricks_replacement"), - notes=list(motif.get("notes") or []), - source_type_hint=motif.get("source_type_hint"), - ) - for motif in raw.get("motifs") or [] - ], - ) - - def _parameter_to_dict(spec: ParameterSpec) -> dict[str, Any]: result: dict[str, Any] = {} if spec.type is not None: diff --git a/src/flowx/ir_serde.py b/src/flowx/ir_serde.py index d539d1a5..3dbbe319 100644 --- a/src/flowx/ir_serde.py +++ b/src/flowx/ir_serde.py @@ -22,8 +22,10 @@ from flowx.models.ir import ( Activity, AppendVariableActivity, + ControlEdge, CopyActivity, DataAsset, + DataEdge, DbtFactoryActivity, DeleteActivity, ExecutePipelineActivity, @@ -150,6 +152,48 @@ def lineage_to_dict(lineage: Lineage) -> dict[str, Any]: } +def lineage_from_dict(raw: dict[str, Any]) -> Lineage: + """Rehydrate a :class:`Lineage` block from the dict :func:`lineage_to_dict` emits. + + The one inverse shared by the translation report and the discovery graph, so a + lineage block reads back the same way wherever it was written. + """ + return Lineage( + control_edges=[ + ControlEdge( + source_workflow=edge.get("source_workflow", ""), + target_workflow=edge.get("target_workflow", ""), + via_task_key=edge.get("via_task_key", ""), + wait_for_completion=edge.get("wait_for_completion"), + resolved=bool(edge.get("resolved", True)), + ) + for edge in raw.get("control_edges") or [] + ], + data_edges=[ + DataEdge( + source_task_key=edge.get("source_task_key", ""), + target_task_key=edge.get("target_task_key", ""), + match_kind=edge.get("match_kind", ""), + match_key=edge.get("match_key", ""), + identity=edge.get("identity"), + asset_type=edge.get("asset_type"), + ) + for edge in raw.get("data_edges") or [] + ], + motifs=[ + MotifAnnotation( + motif_id=motif.get("motif_id", ""), + member_task_keys=list(motif.get("member_task_keys") or []), + display_name=motif.get("display_name"), + databricks_replacement=motif.get("databricks_replacement"), + notes=list(motif.get("notes") or []), + source_type_hint=motif.get("source_type_hint"), + ) + for motif in raw.get("motifs") or [] + ], + ) + + def _motif_annotation_to_dict(motif: MotifAnnotation) -> dict[str, Any]: """Serialise one motif annotation; ``source_type_hint`` is written only when set. diff --git a/src/flowx/sources/adf/dataset_lineage.py b/src/flowx/sources/adf/dataset_lineage.py index ba51aa7b..27926511 100644 --- a/src/flowx/sources/adf/dataset_lineage.py +++ b/src/flowx/sources/adf/dataset_lineage.py @@ -9,8 +9,9 @@ Two tiers, exactly as #36 established them: -* **identity** -- the resolved physical location of the asset (``schema.table`` or - a concrete ``abfss://`` path). Present only when it resolves *deterministically* +* **identity** -- the resolved physical location of the asset (a store-qualified + ``/schema.table`` or a concrete ``abfss://`` path down to the literal + file name). Present only when it resolves *deterministically* from literals; a parameterised reference is never guessed at and leaves ``identity`` unset. This is the strong join key. * **signature** -- the *path-derived* weak key. For an asset with a resolved @@ -204,9 +205,15 @@ def resolve_dataset_identity( ) -> str | None: """Deterministic physical identity for a dataset reference. - Returns ``"schema.table"`` when a table is resolvable, else a storage path, - else ``None`` (never a guess). Used to join producers to consumers on the same - physical asset even when their ADF dataset names differ. + Returns ``"/schema.table"`` when a table is resolvable, else a storage + path down to the literal file name, else ``None`` (never a guess). Used to join + producers to consumers on the same physical asset even when their ADF dataset + names differ. + + A table name is only unique within the store that holds it, so the table is + prefixed with the backing linked service's server (and database, when known), + or with the linked service name when no server literal resolves. Otherwise a + source ``dbo.Orders`` and a warehouse ``dbo.Orders`` would look like one table. Parameterised values (ADF expressions or DAB-ref placeholders) are treated as unresolvable and return ``None`` -- they must never be used as identity keys @@ -217,12 +224,17 @@ def resolve_dataset_identity( properties = _dataset_props(dataset_ref, definitions) if properties is None: return None + linked_service_name, linked_service = _backing_linked_service(properties, definitions) schema, table = _resolve_table_reference(dataset_ref, properties, resolution_context) if table: - identity = f"{schema}.{table}" if schema else table - return identity if _is_physical(identity) else None - path = _resolve_dataset_path(properties, definitions) - return path if (path and _is_physical(path)) else None + if not linked_service_name: + return None + qualified_table = f"{schema}.{table}" if schema else table + if not _is_physical(qualified_table): + return None + store = _resolve_table_store(linked_service) or linked_service_name + return f"{store}/{qualified_table}" + return _resolve_dataset_path(properties, linked_service) def _dataset_props(dataset_ref: AdfDatasetReference, definitions: AdfDefinitions) -> dict[str, Any] | None: @@ -338,8 +350,70 @@ def _pick_dataset_field( return None -def _resolve_dataset_path(dataset_props: dict[str, Any], definitions: AdfDefinitions) -> str | None: - """Resolve a dataset's storage path from its location + backing linked service.""" +def _backing_linked_service(dataset_props: dict[str, Any], definitions: AdfDefinitions) -> tuple[str, Any]: + """Return the dataset's linked service name and its loaded definition (``None`` when not loaded).""" + linked_service_reference = dataset_props.get("linkedServiceName") or {} + if isinstance(linked_service_reference, dict): + linked_service_name = linked_service_reference.get("referenceName") or "" + else: + linked_service_name = str(linked_service_reference) + linked_service = definitions.get_linked_service(linked_service_name) if linked_service_name else None + return linked_service_name, linked_service + + +def _resolve_table_store(linked_service: Any) -> str | None: + """Literal ``server`` or ``server/database`` behind a database linked service, or ``None``. + + Reads the explicit ``server`` / ``database`` properties first and falls back to + a plaintext connection string. A Key Vault reference, a masked secret, or a + parameterised server gives ``None`` so the caller qualifies by linked service + name instead of guessing. + """ + if linked_service is None: + return None + type_props = linked_service.properties.get("typeProperties") or linked_service.properties + connection_string = type_props.get("connectionString") + if isinstance(connection_string, dict): + connection_string = connection_string.get("value") + connection_fields = _connection_string_fields(connection_string) if isinstance(connection_string, str) else {} + + server = type_props.get("server") or connection_fields.get("server") or connection_fields.get("data source") + database = ( + type_props.get("database") or connection_fields.get("database") or connection_fields.get("initial catalog") + ) + if not isinstance(server, str) or not server.strip() or not _is_physical(server): + return None + store = _normalize_server(server) + if isinstance(database, str) and database.strip() and _is_physical(database): + store = f"{store}/{database.strip()}" + return store + + +def _connection_string_fields(connection_string: str) -> dict[str, str]: + """Split a ``key=value;`` connection string into a lower-cased key map.""" + fields: dict[str, str] = {} + for part in connection_string.split(";"): + key, separator, value = part.partition("=") + if separator: + fields[key.strip().lower()] = value.strip() + return fields + + +def _normalize_server(server: str) -> str: + """Drop the ``tcp:`` prefix and port so ``tcp:host,1433`` and ``host`` name the same server.""" + host = server.strip() + if host.lower().startswith("tcp:"): + host = host[len("tcp:") :] + return host.split(",", 1)[0].lower() + + +def _resolve_dataset_path(dataset_props: dict[str, Any], linked_service: Any) -> str | None: + """Resolve a dataset's storage path from its location + backing linked service. + + The path runs down to the literal ``fileName`` when the dataset names one, so + two files in the same folder stay distinct. Any parameterised or unresolved + location part (file system, folder, or file name) yields ``None``. + """ type_props = dataset_props.get("typeProperties") or dataset_props location = type_props.get("location") or {} if not isinstance(location, dict): @@ -347,20 +421,17 @@ def _resolve_dataset_path(dataset_props: dict[str, Any], definitions: AdfDefinit file_system = location.get("fileSystem") or location.get("container") or "" folder_path = location.get("folderPath") or "" - if isinstance(file_system, dict) or isinstance(folder_path, dict): - return None # parameterised location; not a deterministic identity + file_name = location.get("fileName") or "" + location_parts = (file_system, folder_path, file_name) + if not all(isinstance(part, str) and _is_physical(part) for part in location_parts) or not file_system: + return None - linked_service_reference = dataset_props.get("linkedServiceName") or {} - if isinstance(linked_service_reference, dict): - linked_service_name = linked_service_reference.get("referenceName", "") - else: - linked_service_name = str(linked_service_reference) - linked_service = definitions.get_linked_service(linked_service_name) if linked_service_name else None account = _resolve_storage_account(linked_service) if not account: return None - return f"abfss://{file_system}@{account}.dfs.core.windows.net/{folder_path}".rstrip("/") + relative_path = "/".join(part.strip("/") for part in (folder_path, file_name) if part.strip("/")) + return f"abfss://{file_system}@{account}.dfs.core.windows.net/{relative_path}".rstrip("/") def _resolve_storage_account(linked_service: Any) -> str | None: diff --git a/tests/unit/test_adf_dataset_lineage.py b/tests/unit/test_adf_dataset_lineage.py index 8d3f9a54..9cfe2d5a 100644 --- a/tests/unit/test_adf_dataset_lineage.py +++ b/tests/unit/test_adf_dataset_lineage.py @@ -23,20 +23,28 @@ def _definitions(**datasets: AdfDataset) -> AdfDefinitions: return AdfDefinitions(pipelines=[], datasets=dict(datasets)) -def _table_dataset(name: str, *, schema: str, table: str) -> AdfDataset: +def _table_dataset(name: str, *, schema: str, table: str, linked_service: str = "ls_sql") -> AdfDataset: return AdfDataset( name=name, type="AzureSqlTable", - properties={"typeProperties": {"schema": schema, "table": table}}, + properties={ + "typeProperties": {"schema": schema, "table": table}, + "linkedServiceName": {"referenceName": linked_service}, + }, ) -def _adls_dataset(name: str, *, file_system: str, folder_path: str, linked_service: str) -> AdfDataset: +def _adls_dataset( + name: str, *, file_system: str, folder_path: str, linked_service: str, file_name: object = None +) -> AdfDataset: + location: dict = {"fileSystem": file_system, "folderPath": folder_path} + if file_name is not None: + location["fileName"] = file_name return AdfDataset( name=name, type="DelimitedText", properties={ - "typeProperties": {"location": {"fileSystem": file_system, "folderPath": folder_path}}, + "typeProperties": {"location": location}, "linkedServiceName": {"referenceName": linked_service}, }, ) @@ -50,6 +58,22 @@ def _adls_linked_service(name: str, *, account: str) -> AdfLinkedService: ) +def _sql_linked_service(name: str, *, connection_string: object) -> AdfLinkedService: + return AdfLinkedService( + name=name, + type="AzureSqlDatabase", + properties={"typeProperties": {"connectionString": connection_string}}, + ) + + +def _copy(name: str, *, source: str, sink: str) -> AdfActivity: + return AdfActivity( + name=name, + type="Copy", + type_properties={"source": {"referenceName": source}, "sink": {"referenceName": sink}}, + ) + + # --------------------------------------------------------------------------- # # Identity tier # --------------------------------------------------------------------------- # @@ -58,7 +82,7 @@ def _adls_linked_service(name: str, *, account: str) -> AdfLinkedService: def test_identity_resolves_schema_and_table() -> None: definitions = _definitions(ds_orders=_table_dataset("ds_orders", schema="curated", table="orders")) identity = resolve_dataset_identity(AdfDatasetReference(reference_name="ds_orders"), definitions) - assert identity == "curated.orders" + assert identity == "ls_sql/curated.orders" def test_identity_resolves_storage_path_from_linked_service() -> None: @@ -73,6 +97,117 @@ def test_identity_resolves_storage_path_from_linked_service() -> None: assert identity == "abfss://data@contosolake.dfs.core.windows.net/raw/customers" +def test_files_in_the_same_folder_resolve_to_distinct_identities() -> None: + """``raw/orders.csv`` and ``raw/customers.csv`` are different assets, so no identity edge joins them.""" + definitions = AdfDefinitions( + pipelines=[], + datasets={ + "ds_orders": _adls_dataset( + "ds_orders", file_system="data", folder_path="raw", linked_service="ls", file_name="orders.csv" + ), + "ds_customers": _adls_dataset( + "ds_customers", file_system="data", folder_path="raw", linked_service="ls", file_name="customers.csv" + ), + "ds_orders_again": _adls_dataset( + "ds_orders_again", file_system="data", folder_path="raw", linked_service="ls", file_name="orders.csv" + ), + }, + linked_services={"ls": _adls_linked_service("ls", account="acct")}, + ) + _, copy1_writes = activity_data_assets(_copy("Copy1", source="ds_orders_again", sink="ds_orders"), definitions) + copy2_reads, _ = activity_data_assets(_copy("Copy2", source="ds_customers", sink="ds_orders_again"), definitions) + + assert copy1_writes[0].identity == "abfss://data@acct.dfs.core.windows.net/raw/orders.csv" + assert copy2_reads[0].identity == "abfss://data@acct.dfs.core.windows.net/raw/customers.csv" + assert copy1_writes[0].signature != copy2_reads[0].signature + same_file_identity = resolve_dataset_identity(AdfDatasetReference(reference_name="ds_orders_again"), definitions) + assert same_file_identity == copy1_writes[0].identity + + +def test_parameterised_file_name_has_no_path_identity() -> None: + """A per-entity file name is only known at run time, so the folder alone is never used as its identity.""" + definitions = AdfDefinitions( + pipelines=[], + datasets={ + "ds_expression": _adls_dataset( + "ds_expression", + file_system="data", + folder_path="raw", + linked_service="ls", + file_name={"value": "@dataset().entity", "type": "Expression"}, + ), + "ds_bare_expression": _adls_dataset( + "ds_bare_expression", + file_system="data", + folder_path="raw", + linked_service="ls", + file_name="@dataset().entity", + ), + }, + linked_services={"ls": _adls_linked_service("ls", account="acct")}, + ) + assert resolve_dataset_identity(AdfDatasetReference(reference_name="ds_expression"), definitions) is None + assert resolve_dataset_identity(AdfDatasetReference(reference_name="ds_bare_expression"), definitions) is None + + +def test_same_table_name_on_different_servers_resolves_to_distinct_identities() -> None: + """A source ``dbo.Orders`` and a warehouse ``dbo.Orders`` are different tables, so no identity edge joins them.""" + definitions = AdfDefinitions( + pipelines=[], + datasets={ + "ds_source_orders": _table_dataset( + "ds_source_orders", schema="dbo", table="Orders", linked_service="ls_src" + ), + "ds_warehouse_orders": _table_dataset( + "ds_warehouse_orders", schema="dbo", table="Orders", linked_service="ls_dw" + ), + "ds_lake": _adls_dataset("ds_lake", file_system="lake", folder_path="orders", linked_service="ls_lake"), + }, + linked_services={ + "ls_src": _sql_linked_service( + "ls_src", connection_string="Server=tcp:onprem-sql.contoso.local,1433;Database=Sales;User ID=etl" + ), + "ls_dw": _sql_linked_service( + "ls_dw", connection_string={"type": "SecureString", "value": "Data Source=dw.contoso.net;"} + ), + "ls_lake": _adls_linked_service("ls_lake", account="lake"), + }, + ) + copy_a_reads, _ = activity_data_assets(_copy("Copy A", source="ds_source_orders", sink="ds_lake"), definitions) + _, copy_b_writes = activity_data_assets(_copy("Copy B", source="ds_lake", sink="ds_warehouse_orders"), definitions) + + assert copy_a_reads[0].identity == "onprem-sql.contoso.local/Sales/dbo.Orders" + assert copy_b_writes[0].identity == "dw.contoso.net/dbo.Orders" + assert copy_a_reads[0].signature != copy_b_writes[0].signature + + +def test_table_identity_falls_back_to_linked_service_name_when_server_is_secret() -> None: + """A Key Vault connection string hides the server, so the linked service name qualifies the table.""" + key_vault_connection = { + "type": "AzureKeyVaultSecret", + "store": {"referenceName": "ls_key_vault", "type": "LinkedServiceReference"}, + "secretName": "sql-connection", + } + definitions = AdfDefinitions( + pipelines=[], + datasets={"ds_orders": _table_dataset("ds_orders", schema="dbo", table="Orders", linked_service="ls_vaulted")}, + linked_services={"ls_vaulted": _sql_linked_service("ls_vaulted", connection_string=key_vault_connection)}, + ) + identity = resolve_dataset_identity(AdfDatasetReference(reference_name="ds_orders"), definitions) + assert identity == "ls_vaulted/dbo.Orders" + + +def test_table_without_linked_service_has_no_identity() -> None: + """With no backing store named, a bare ``schema.table`` is not provably one physical table.""" + dataset = AdfDataset( + name="ds_orphan", + type="AzureSqlTable", + properties={"typeProperties": {"schema": "dbo", "table": "Orders"}}, + ) + definitions = _definitions(ds_orphan=dataset) + assert resolve_dataset_identity(AdfDatasetReference(reference_name="ds_orphan"), definitions) is None + + def test_identity_resolves_dataset_param_from_call_site_literal() -> None: """A ``@dataset().table`` expression resolves when the call site passes a literal.""" dataset = AdfDataset( @@ -81,11 +216,12 @@ def test_identity_resolves_dataset_param_from_call_site_literal() -> None: properties={ "typeProperties": {"schema": "dbo", "table": "@dataset().tbl"}, "parameters": {"tbl": {"type": "String"}}, + "linkedServiceName": {"referenceName": "ls_sql"}, }, ) definitions = _definitions(ds_param=dataset) reference = AdfDatasetReference(reference_name="ds_param", parameters={"tbl": "shipments"}) - assert resolve_dataset_identity(reference, definitions) == "dbo.shipments" + assert resolve_dataset_identity(reference, definitions) == "ls_sql/dbo.shipments" def test_parameterised_table_is_not_guessed() -> None: @@ -93,7 +229,10 @@ def test_parameterised_table_is_not_guessed() -> None: dataset = AdfDataset( name="ds_dyn", type="AzureSqlTable", - properties={"typeProperties": {"schema": "dbo", "table": "@pipeline().parameters.tableName"}}, + properties={ + "typeProperties": {"schema": "dbo", "table": "@pipeline().parameters.tableName"}, + "linkedServiceName": {"referenceName": "ls_sql"}, + }, ) definitions = _definitions(ds_dyn=dataset) assert resolve_dataset_identity(AdfDatasetReference(reference_name="ds_dyn"), definitions) is None @@ -122,8 +261,10 @@ def test_copy_captures_source_read_and_sink_write_with_identity() -> None: }, ) reads, writes = activity_data_assets(activity, definitions) - assert [(asset.identity, asset.signature) for asset in reads] == [("raw.orders", "raw.orders")] - assert [(asset.identity, asset.signature) for asset in writes] == [("curated.orders", "curated.orders")] + assert [(asset.identity, asset.signature) for asset in reads] == [("ls_sql/raw.orders", "ls_sql/raw.orders")] + assert [(asset.identity, asset.signature) for asset in writes] == [ + ("ls_sql/curated.orders", "ls_sql/curated.orders") + ] assert reads[0].asset_type == "table" @@ -142,8 +283,8 @@ def test_captures_all_inputs_and_outputs_not_just_index_zero() -> None: outputs=[AdfDatasetReference(reference_name="ds_out_a"), AdfDatasetReference(reference_name="ds_out_b")], ) reads, writes = activity_data_assets(activity, definitions) - assert sorted(asset.identity for asset in reads) == ["raw.a", "raw.b"] - assert sorted(asset.identity for asset in writes) == ["curated.a", "curated.b"] + assert sorted(asset.identity for asset in reads) == ["ls_sql/raw.a", "ls_sql/raw.b"] + assert sorted(asset.identity for asset in writes) == ["ls_sql/curated.a", "ls_sql/curated.b"] def test_dataset_named_in_both_slot_and_typeproperties_counted_once() -> None: @@ -155,7 +296,7 @@ def test_dataset_named_in_both_slot_and_typeproperties_counted_once() -> None: type_properties={"dataset": {"referenceName": "ds_src"}}, ) reads, _ = activity_data_assets(activity, definitions) - assert [asset.identity for asset in reads] == ["raw.orders"] + assert [asset.identity for asset in reads] == ["ls_sql/raw.orders"] def test_unresolved_reference_falls_back_to_path_signature() -> None: @@ -203,6 +344,7 @@ def test_distinct_parameter_bindings_of_same_dataset_are_all_retained() -> None: properties={ "typeProperties": {"schema": "raw", "table": "@dataset().tbl"}, "parameters": {"tbl": {"type": "String"}}, + "linkedServiceName": {"referenceName": "ls_sql"}, }, ) definitions = _definitions(ds=dataset) @@ -215,7 +357,7 @@ def test_distinct_parameter_bindings_of_same_dataset_are_all_retained() -> None: ], ) reads, _ = activity_data_assets(activity, definitions) - assert sorted(asset.identity for asset in reads) == ["raw.customers", "raw.orders"] + assert sorted(asset.identity for asset in reads) == ["ls_sql/raw.customers", "ls_sql/raw.orders"] def test_same_ref_same_params_in_slot_and_typeproperties_still_collapses() -> None: @@ -228,7 +370,7 @@ def test_same_ref_same_params_in_slot_and_typeproperties_still_collapses() -> No type_properties={"dataset": {"referenceName": "ds_src"}}, ) reads, _ = activity_data_assets(activity, definitions) - assert [asset.identity for asset in reads] == ["raw.orders"] + assert [asset.identity for asset in reads] == ["ls_sql/raw.orders"] def _ref_dict(reference: AdfDatasetReference) -> dict: diff --git a/tests/unit/test_lineage_substrate.py b/tests/unit/test_lineage_substrate.py index 1f1d8a8c..1bdb02ee 100644 --- a/tests/unit/test_lineage_substrate.py +++ b/tests/unit/test_lineage_substrate.py @@ -396,6 +396,7 @@ def test_serde_round_trip_new_activity_fields_and_lineage_block(): display_name="Auto Loader", databricks_replacement="auto_loader", notes=["note"], + source_type_hint="AzureBlobFSReadSettings", ) ], ), From 1068dceb234093a9980a51a4a20519aa5ea07b7f Mon Sep 17 00:00:00 2001 From: Matthew Moorcroft Date: Tue, 6 Oct 2026 16:18:03 +0100 Subject: [PATCH 08/25] no-mistakes(review): Reject parameterised linked-service parts in ADF dataset identities --- src/flowx/sources/adf/dataset_lineage.py | 93 +++++++++++----------- tests/unit/test_adf_dataset_lineage.py | 99 +++++++++++++++++++++++- 2 files changed, 146 insertions(+), 46 deletions(-) diff --git a/src/flowx/sources/adf/dataset_lineage.py b/src/flowx/sources/adf/dataset_lineage.py index 27926511..ff9312f5 100644 --- a/src/flowx/sources/adf/dataset_lineage.py +++ b/src/flowx/sources/adf/dataset_lineage.py @@ -211,30 +211,28 @@ def resolve_dataset_identity( names differ. A table name is only unique within the store that holds it, so the table is - prefixed with the backing linked service's server (and database, when known), - or with the linked service name when no server literal resolves. Otherwise a - source ``dbo.Orders`` and a warehouse ``dbo.Orders`` would look like one table. - - Parameterised values (ADF expressions or DAB-ref placeholders) are treated as - unresolvable and return ``None`` -- they must never be used as identity keys - because two unrelated pipelines sharing a parameter name would collide on the - same placeholder string. + prefixed with the backing linked service's server (and database, when given), + or with the linked service name when the server is hidden in a secret. + Otherwise a source ``dbo.Orders`` and a warehouse ``dbo.Orders`` would look like + one table. + + Parameterised values (ADF expressions or DAB-ref placeholders), including a + parameterised linked service, are treated as unresolvable and return ``None`` + -- they must never be used as identity keys because two unrelated pipelines + sharing a parameter name would collide on the same placeholder string. """ resolution_context = context if context is not None else TranslationContext() properties = _dataset_props(dataset_ref, definitions) if properties is None: return None - linked_service_name, linked_service = _backing_linked_service(properties, definitions) schema, table = _resolve_table_reference(dataset_ref, properties, resolution_context) if table: - if not linked_service_name: - return None qualified_table = f"{schema}.{table}" if schema else table if not _is_physical(qualified_table): return None - store = _resolve_table_store(linked_service) or linked_service_name - return f"{store}/{qualified_table}" - return _resolve_dataset_path(properties, linked_service) + store = _resolve_table_store(properties, definitions) + return f"{store}/{qualified_table}" if store else None + return _resolve_dataset_path(properties, _backing_linked_service(properties, definitions)) def _dataset_props(dataset_ref: AdfDatasetReference, definitions: AdfDefinitions) -> dict[str, Any] | None: @@ -350,43 +348,56 @@ def _pick_dataset_field( return None -def _backing_linked_service(dataset_props: dict[str, Any], definitions: AdfDefinitions) -> tuple[str, Any]: - """Return the dataset's linked service name and its loaded definition (``None`` when not loaded).""" +def _linked_service_reference(dataset_props: dict[str, Any]) -> tuple[str, dict[str, Any]]: + """Return the dataset's linked service name and the parameters its reference binds.""" linked_service_reference = dataset_props.get("linkedServiceName") or {} - if isinstance(linked_service_reference, dict): - linked_service_name = linked_service_reference.get("referenceName") or "" - else: - linked_service_name = str(linked_service_reference) - linked_service = definitions.get_linked_service(linked_service_name) if linked_service_name else None - return linked_service_name, linked_service + if not isinstance(linked_service_reference, dict): + return str(linked_service_reference), {} + parameters = linked_service_reference.get("parameters") + return linked_service_reference.get("referenceName") or "", parameters if isinstance(parameters, dict) else {} + +def _backing_linked_service(dataset_props: dict[str, Any], definitions: AdfDefinitions) -> Any: + """Return the loaded definition of the dataset's linked service, or ``None``.""" + linked_service_name, _ = _linked_service_reference(dataset_props) + return definitions.get_linked_service(linked_service_name) if linked_service_name else None -def _resolve_table_store(linked_service: Any) -> str | None: - """Literal ``server`` or ``server/database`` behind a database linked service, or ``None``. - Reads the explicit ``server`` / ``database`` properties first and falls back to - a plaintext connection string. A Key Vault reference, a masked secret, or a - parameterised server gives ``None`` so the caller qualifies by linked service - name instead of guessing. +def _resolve_table_store(dataset_props: dict[str, Any], definitions: AdfDefinitions) -> str | None: + """Name the store that holds a dataset's table, or ``None`` when it is not provable. + + Uses the literal ``server`` (plus ``database`` when given) from the backing + linked service's properties or plaintext connection string, exactly as written. + When the server is hidden (a Key Vault reference or a masked secret) the linked + service name stands in for it. A parameterised server, database, or connection + string, or a linked service that takes parameters, gives ``None``: each binding + may point at a different store, so no single name identifies it. """ - if linked_service is None: + linked_service_name, reference_parameters = _linked_service_reference(dataset_props) + if not linked_service_name: return None - type_props = linked_service.properties.get("typeProperties") or linked_service.properties + linked_service = definitions.get_linked_service(linked_service_name) + linked_service_properties = linked_service.properties if linked_service is not None else {} + type_props = linked_service_properties.get("typeProperties") or linked_service_properties connection_string = type_props.get("connectionString") if isinstance(connection_string, dict): connection_string = connection_string.get("value") + if isinstance(connection_string, str) and not _is_physical(connection_string): + return None connection_fields = _connection_string_fields(connection_string) if isinstance(connection_string, str) else {} server = type_props.get("server") or connection_fields.get("server") or connection_fields.get("data source") database = ( type_props.get("database") or connection_fields.get("database") or connection_fields.get("initial catalog") ) - if not isinstance(server, str) or not server.strip() or not _is_physical(server): + store_parts = [part for part in (server, database) if part] + if not all(isinstance(part, str) and part.strip() and _is_physical(part) for part in store_parts): + return None + if server: + return "/".join(part.strip() for part in store_parts) + if reference_parameters or linked_service_properties.get("parameters"): return None - store = _normalize_server(server) - if isinstance(database, str) and database.strip() and _is_physical(database): - store = f"{store}/{database.strip()}" - return store + return linked_service_name def _connection_string_fields(connection_string: str) -> dict[str, str]: @@ -399,20 +410,12 @@ def _connection_string_fields(connection_string: str) -> dict[str, str]: return fields -def _normalize_server(server: str) -> str: - """Drop the ``tcp:`` prefix and port so ``tcp:host,1433`` and ``host`` name the same server.""" - host = server.strip() - if host.lower().startswith("tcp:"): - host = host[len("tcp:") :] - return host.split(",", 1)[0].lower() - - def _resolve_dataset_path(dataset_props: dict[str, Any], linked_service: Any) -> str | None: """Resolve a dataset's storage path from its location + backing linked service. The path runs down to the literal ``fileName`` when the dataset names one, so two files in the same folder stay distinct. Any parameterised or unresolved - location part (file system, folder, or file name) yields ``None``. + part (file system, folder, file name, or storage account) yields ``None``. """ type_props = dataset_props.get("typeProperties") or dataset_props location = type_props.get("location") or {} @@ -427,7 +430,7 @@ def _resolve_dataset_path(dataset_props: dict[str, Any], linked_service: Any) -> return None account = _resolve_storage_account(linked_service) - if not account: + if not account or not _is_physical(account): return None relative_path = "/".join(part.strip("/") for part in (folder_path, file_name) if part.strip("/")) diff --git a/tests/unit/test_adf_dataset_lineage.py b/tests/unit/test_adf_dataset_lineage.py index 9cfe2d5a..cd7d8476 100644 --- a/tests/unit/test_adf_dataset_lineage.py +++ b/tests/unit/test_adf_dataset_lineage.py @@ -176,7 +176,7 @@ def test_same_table_name_on_different_servers_resolves_to_distinct_identities() copy_a_reads, _ = activity_data_assets(_copy("Copy A", source="ds_source_orders", sink="ds_lake"), definitions) _, copy_b_writes = activity_data_assets(_copy("Copy B", source="ds_lake", sink="ds_warehouse_orders"), definitions) - assert copy_a_reads[0].identity == "onprem-sql.contoso.local/Sales/dbo.Orders" + assert copy_a_reads[0].identity == "tcp:onprem-sql.contoso.local,1433/Sales/dbo.Orders" assert copy_b_writes[0].identity == "dw.contoso.net/dbo.Orders" assert copy_a_reads[0].signature != copy_b_writes[0].signature @@ -197,6 +197,103 @@ def test_table_identity_falls_back_to_linked_service_name_when_server_is_secret( assert identity == "ls_vaulted/dbo.Orders" +def test_parameterised_storage_account_has_no_path_identity() -> None: + """A generic ADLS linked service names its account per binding, so the account is never part of an identity.""" + generic_lake = AdfLinkedService( + name="ls_generic_lake", + type="AzureBlobFS", + properties={ + "typeProperties": {"url": "https://@{linkedService().accountName}.dfs.core.windows.net"}, + "parameters": {"accountName": {"type": "String"}}, + }, + ) + definitions = AdfDefinitions( + pipelines=[], + datasets={ + "ds_orders": _adls_dataset( + "ds_orders", + file_system="data", + folder_path="raw", + linked_service="ls_generic_lake", + file_name="orders.csv", + ) + }, + linked_services={"ls_generic_lake": generic_lake}, + ) + assert resolve_dataset_identity(AdfDatasetReference(reference_name="ds_orders"), definitions) is None + + +def test_parameterised_database_has_no_table_identity() -> None: + """A literal server with a parameterised database does not say which database holds the table.""" + definitions = AdfDefinitions( + pipelines=[], + datasets={ + "ds_property": _table_dataset("ds_property", schema="dbo", table="Orders", linked_service="ls_property"), + "ds_connection": _table_dataset( + "ds_connection", schema="dbo", table="Orders", linked_service="ls_connection" + ), + }, + linked_services={ + "ls_property": AdfLinkedService( + name="ls_property", + type="SqlServer", + properties={"typeProperties": {"server": "sql.contoso.net", "database": "@{linkedService().dbName}"}}, + ), + "ls_connection": _sql_linked_service( + "ls_connection", connection_string="Server=sql.contoso.net;Initial Catalog=@{linkedService().dbName}" + ), + }, + ) + assert resolve_dataset_identity(AdfDatasetReference(reference_name="ds_property"), definitions) is None + assert resolve_dataset_identity(AdfDatasetReference(reference_name="ds_connection"), definitions) is None + + +def test_generic_linked_service_called_with_different_bindings_has_no_table_identity() -> None: + """One generic SQL linked service bound to a source and a warehouse server never yields one shared identity.""" + generic_sql = AdfLinkedService( + name="ls_generic_sql", + type="AzureSqlDatabase", + properties={ + "typeProperties": { + "connectionString": { + "type": "AzureKeyVaultSecret", + "store": {"referenceName": "ls_key_vault", "type": "LinkedServiceReference"}, + "secretName": "@{linkedService().secretName}", + } + }, + "parameters": {"secretName": {"type": "String"}}, + }, + ) + + def _bound_orders(name: str, secret_name: str) -> AdfDataset: + return AdfDataset( + name=name, + type="AzureSqlTable", + properties={ + "typeProperties": {"schema": "dbo", "table": "Orders"}, + "linkedServiceName": { + "referenceName": "ls_generic_sql", + "parameters": {"secretName": secret_name}, + }, + }, + ) + + definitions = AdfDefinitions( + pipelines=[], + datasets={ + "ds_source_orders": _bound_orders("ds_source_orders", "source-sql"), + "ds_warehouse_orders": _bound_orders("ds_warehouse_orders", "warehouse-sql"), + }, + linked_services={"ls_generic_sql": generic_sql}, + ) + copy_reads, copy_writes = activity_data_assets( + _copy("Copy", source="ds_source_orders", sink="ds_warehouse_orders"), definitions + ) + + assert copy_reads[0].identity is None + assert copy_writes[0].identity is None + + def test_table_without_linked_service_has_no_identity() -> None: """With no backing store named, a bare ``schema.table`` is not provably one physical table.""" dataset = AdfDataset( From e50cf0f24d8ca695e3a4fc8f0a95e1f18dc37c92 Mon Sep 17 00:00:00 2001 From: Matthew Moorcroft Date: Tue, 6 Oct 2026 16:33:17 +0100 Subject: [PATCH 09/25] no-mistakes(document): Document ADF source_graphs.json and inventory hash fields --- AGENTS.md | 2 +- README.md | 1 + skills/flowx-discover/SKILL.md | 1 + skills/flowx-discover/sources/adf.md | 10 +++++++--- 4 files changed, 10 insertions(+), 4 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index e1ae8c40..8befda10 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -56,7 +56,7 @@ All three phases write into one shared `` (default `./flowx_output`) the DAB bundle at the top level, kept artifacts under `metadata/`, and transient intermediates under `.work/` (pruned by `package`). -1. **Discover** -- Parse ADF JSON from UC volumes -> typed AST -> `metadata/inventory.json` + `metadata/profile_report.csv` + verbatim `metadata/.arm.json` +1. **Discover** -- Parse ADF JSON from UC volumes -> typed AST -> `metadata/source_graphs.json` (hashed) -> `metadata/inventory.json` + `metadata/profile_report.csv` + verbatim `metadata/.arm.json` 2. **Convert** -- Registry dispatch + topological sort -> Pipeline IR (deterministic + agentic gaps); transient report at `.work/translation_report.json` 3. **Package** -- IR -> DAB YAML + generated notebooks + setup scripts; prunes `.work/` diff --git a/README.md b/README.md index ea3e7e8e..42d016a8 100644 --- a/README.md +++ b/README.md @@ -257,6 +257,7 @@ flowx_output/ SETUP.md # Setup instructions (package) metadata/ inventory.json # discover: activity inventory + source_graphs.json # discover (ADF): saved source graphs the inventory is built from profile_report.csv # discover: per-pipeline complexity report .arm.json # discover: verbatim original ADF/ARM source configuration.json # modify: collected configuration answers diff --git a/skills/flowx-discover/SKILL.md b/skills/flowx-discover/SKILL.md index b27e6f06..6c3e9b54 100644 --- a/skills/flowx-discover/SKILL.md +++ b/skills/flowx-discover/SKILL.md @@ -65,6 +65,7 @@ All under the shared `/metadata/` folder: | File | Description | |---|---| | `metadata/inventory.json` | Classified activity inventory for the convert phase | +| `metadata/source_graphs.json` | (ADF) Saved, content-hashed source graphs (activities, lineage, detected motifs); the inventory is built from it and records its hash as `source_graphs_sha256` | | `metadata/profile_report.csv` | Per-pipeline complexity report (counts + T-shirt size) | | `metadata/.arm.json` | (ADF) Verbatim original source for each pipeline (provenance) | diff --git a/skills/flowx-discover/sources/adf.md b/skills/flowx-discover/sources/adf.md index 9997e27d..47b74afc 100644 --- a/skills/flowx-discover/sources/adf.md +++ b/skills/flowx-discover/sources/adf.md @@ -51,16 +51,20 @@ Read `/metadata/inventory.json`: { "name": "PipelineName", "activities": [ - {"name": "CopyFromBlob", "type": "Copy", "strategy": "deterministic", "translator": "copy.py"}, - {"name": "RunDataFlow", "type": "ExecuteDataFlow", "strategy": "agentic"} + {"name": "CopyFromBlob", "type": "Copy", "strategy": "deterministic", "task_key": "CopyFromBlob"}, + {"name": "RunDataFlow", "type": "ExecuteDataFlow", "strategy": "agentic", "task_key": "RunDataFlow"} ] } ], "summary": {"pipeline_count": 12, "activity_count": 47, "deterministic_count": 35, - "agentic_count": 10, "unsupported_count": 2, "coverage_pct": 95.7} + "agentic_count": 10, "unsupported_count": 2, "coverage_pct": 95.7}, + "source_graphs_sha256": "" } ``` +Each activity's `task_key` equals its ADF activity name. `source_graphs_sha256` names the saved +`metadata/source_graphs.json` the inventory was built from. + ## Step 4b — Review the complexity report `/metadata/profile_report.csv` has one row per pipeline: `pipeline`, `activities`, From 1ded91c0392a6b40b15db6ded00c718b3a1eb77a Mon Sep 17 00:00:00 2001 From: Matthew Moorcroft Date: Wed, 7 Oct 2026 15:12:15 +0100 Subject: [PATCH 10/25] Tighten ADF lineage identities and skip external run-now edges Unevaluated ARM template expressions ([parameters(...)], [concat(...)]) are now treated as unresolved, so they never become dataset identities or false data edges. File locations written as @dataset().x resolve against the call-site bindings, as table names already did, so parameterised file datasets keep their valid edges; an unbound part still gives no identity. A run-now of an existing job by ID is no longer emitted as a cross-workflow control edge to the caller's own task key. The ADF discover docs record that inventory.json deliberately has no generated_at timestamp. Each fix has a regression test that fails on the previous code. Co-authored-by: Isaac --- skills/flowx-discover/sources/adf.md | 4 ++ src/flowx/lineage.py | 4 +- src/flowx/sources/adf/dataset_lineage.py | 54 ++++++++++++++++++------ tests/unit/test_adf_dataset_lineage.py | 50 ++++++++++++++++++++++ tests/unit/test_lineage_substrate.py | 10 +++++ 5 files changed, 107 insertions(+), 15 deletions(-) diff --git a/skills/flowx-discover/sources/adf.md b/skills/flowx-discover/sources/adf.md index 47b74afc..a188db40 100644 --- a/skills/flowx-discover/sources/adf.md +++ b/skills/flowx-discover/sources/adf.md @@ -65,6 +65,10 @@ Read `/metadata/inventory.json`: Each activity's `task_key` equals its ADF activity name. `source_graphs_sha256` names the saved `metadata/source_graphs.json` the inventory was built from. +The inventory deliberately has no top-level `generated_at` timestamp any more, so the same export +always produces the same `inventory.json` bytes and the hashes that bind enrich and route to it stay +stable. Use the file's modification time if you need to know when discover ran. + ## Step 4b — Review the complexity report `/metadata/profile_report.csv` has one row per pipeline: `pipeline`, `activities`, diff --git a/src/flowx/lineage.py b/src/flowx/lineage.py index a3edf92b..f05533e3 100644 --- a/src/flowx/lineage.py +++ b/src/flowx/lineage.py @@ -140,7 +140,9 @@ def _calls() -> Iterator[tuple[str, bool | None, str]]: match activity: case ExecutePipelineActivity(): yield activity.pipeline_name or "", activity.wait_on_completion, activity.task_key - case RunJobActivity(): + case RunJobActivity() if not activity.existing_job_id: + # A run-now of an existing job by ID targets a job outside this source, and its + # job_name is the caller's own task key, so it is not a cross-workflow edge. yield activity.job_name or "", None, activity.task_key return control_edges_from_calls(pipeline.name, _calls()) diff --git a/src/flowx/sources/adf/dataset_lineage.py b/src/flowx/sources/adf/dataset_lineage.py index ff9312f5..a0ce7806 100644 --- a/src/flowx/sources/adf/dataset_lineage.py +++ b/src/flowx/sources/adf/dataset_lineage.py @@ -227,12 +227,17 @@ def resolve_dataset_identity( return None schema, table = _resolve_table_reference(dataset_ref, properties, resolution_context) if table: - qualified_table = f"{schema}.{table}" if schema else table - if not _is_physical(qualified_table): + if not all(_is_physical(part) for part in (schema, table) if part): return None + qualified_table = f"{schema}.{table}" if schema else table store = _resolve_table_store(properties, definitions) return f"{store}/{qualified_table}" if store else None - return _resolve_dataset_path(properties, _backing_linked_service(properties, definitions)) + return _resolve_dataset_path( + properties, + _backing_linked_service(properties, definitions), + _effective_dataset_params(dataset_ref, properties), + resolution_context, + ) def _dataset_props(dataset_ref: AdfDatasetReference, definitions: AdfDefinitions) -> dict[str, Any] | None: @@ -410,23 +415,39 @@ def _connection_string_fields(connection_string: str) -> dict[str, str]: return fields -def _resolve_dataset_path(dataset_props: dict[str, Any], linked_service: Any) -> str | None: +def _resolve_dataset_path( + dataset_props: dict[str, Any], + linked_service: Any, + dataset_params: dict[str, Any], + context: TranslationContext, +) -> str | None: """Resolve a dataset's storage path from its location + backing linked service. The path runs down to the literal ``fileName`` when the dataset names one, so - two files in the same folder stay distinct. Any parameterised or unresolved - part (file system, folder, file name, or storage account) yields ``None``. + two files in the same folder stay distinct. Location parts written as + ``@dataset().x`` resolve against the reference's effective parameters, as table + names do, so a parameterised dataset bound to literals still gets its identity. + Any part that stays parameterised or unresolved (file system, folder, file name, + or storage account) yields ``None``. """ type_props = dataset_props.get("typeProperties") or dataset_props location = type_props.get("location") or {} if not isinstance(location, dict): return None - file_system = location.get("fileSystem") or location.get("container") or "" - folder_path = location.get("folderPath") or "" - file_name = location.get("fileName") or "" - location_parts = (file_system, folder_path, file_name) - if not all(isinstance(part, str) and _is_physical(part) for part in location_parts) or not file_system: + def _resolved_part(raw: Any) -> str | None: + if raw is None or raw == "": + return "" + resolved = _resolve_param_value(raw, dataset_params, context) + # A part that was written but resolves to nothing (an unbound parameter) is unknown, not absent. + if not resolved or not _is_physical(resolved): + return None + return resolved + + file_system = _resolved_part(location.get("fileSystem") or location.get("container")) + folder_path = _resolved_part(location.get("folderPath")) + file_name = _resolved_part(location.get("fileName")) + if file_system is None or folder_path is None or file_name is None or not file_system: return None account = _resolve_storage_account(linked_service) @@ -478,10 +499,15 @@ def _is_physical(value: str) -> bool: A value is NOT physical when it still contains an unresolved marker -- a DAB-ref placeholder (``{{`` ... ``}}``), a leftover ADF interpolation - fragment (``@{``), or a bare ADF expression (starts with ``@``). + fragment (``@{``), a bare ADF expression (starts with ``@``), or an ARM + template expression the export left unevaluated (``[parameters('x')]``, + ``[concat(...)]``). ARM writes a literal that starts with ``[`` as ``[[``. """ - stripped = value.lstrip() - return not ("{{" in value or "@{" in value or stripped.startswith("@")) + stripped = value.strip() + if "{{" in value or "@{" in value or stripped.startswith("@"): + return False + is_arm_expression = stripped.startswith("[") and not stripped.startswith("[[") and stripped.endswith("]") + return not is_arm_expression # --------------------------------------------------------------------------- # diff --git a/tests/unit/test_adf_dataset_lineage.py b/tests/unit/test_adf_dataset_lineage.py index cd7d8476..22923271 100644 --- a/tests/unit/test_adf_dataset_lineage.py +++ b/tests/unit/test_adf_dataset_lineage.py @@ -476,3 +476,53 @@ def _ref_dict(reference: AdfDatasetReference) -> dict: if reference.parameters: payload["parameters"] = reference.parameters return payload + + +def test_unevaluated_arm_expressions_never_become_identities() -> None: + """An ARM template expression the export left unevaluated is unresolved, so it never forms an identity.""" + definitions = AdfDefinitions( + pipelines=[], + datasets={ + "ds_table": _table_dataset("ds_table", schema="[parameters('schemaName')]", table="Orders"), + "ds_file": _adls_dataset( + "ds_file", file_system="[concat('raw', parameters('env'))]", folder_path="in", linked_service="ls" + ), + "ds_escaped": _adls_dataset("ds_escaped", file_system="[[literal", folder_path="in", linked_service="ls"), + }, + linked_services={ + "ls_sql": AdfLinkedService( + name="ls_sql", type="AzureSqlDatabase", properties={"typeProperties": {"server": "sql.example.net"}} + ), + "ls": _adls_linked_service("ls", account="acct"), + }, + ) + assert resolve_dataset_identity(AdfDatasetReference(reference_name="ds_table"), definitions) is None + assert resolve_dataset_identity(AdfDatasetReference(reference_name="ds_file"), definitions) is None + # ARM's "[[" escape is a literal value that starts with "[", so it stays physical. + assert resolve_dataset_identity(AdfDatasetReference(reference_name="ds_escaped"), definitions) is not None + + +def test_parameterised_file_location_bound_to_literals_resolves_to_its_path() -> None: + """``@dataset().x`` location parts resolve against the call-site bindings, as table names do.""" + definitions = AdfDefinitions( + pipelines=[], + datasets={ + "ds_param": _adls_dataset( + "ds_param", + file_system="data", + folder_path={"value": "@dataset().folder", "type": "Expression"}, + linked_service="ls", + file_name="@dataset().entity", + ), + }, + linked_services={"ls": _adls_linked_service("ls", account="acct")}, + ) + orders = AdfDatasetReference(reference_name="ds_param", parameters={"folder": "raw", "entity": "orders.csv"}) + customers = AdfDatasetReference(reference_name="ds_param", parameters={"folder": "raw", "entity": "customers.csv"}) + unbound = AdfDatasetReference(reference_name="ds_param", parameters={"folder": "raw"}) + assert resolve_dataset_identity(orders, definitions) == "abfss://data@acct.dfs.core.windows.net/raw/orders.csv" + assert ( + resolve_dataset_identity(customers, definitions) == "abfss://data@acct.dfs.core.windows.net/raw/customers.csv" + ) + # An unbound file-name parameter is unknown, so the folder alone is never used as the identity. + assert resolve_dataset_identity(unbound, definitions) is None diff --git a/tests/unit/test_lineage_substrate.py b/tests/unit/test_lineage_substrate.py index 1bdb02ee..b2a9c16e 100644 --- a/tests/unit/test_lineage_substrate.py +++ b/tests/unit/test_lineage_substrate.py @@ -123,6 +123,16 @@ def test_control_edges_run_job_activity_is_source_neutral(): assert edges[0].resolved is True +def test_run_now_of_an_existing_job_is_not_a_cross_workflow_edge(): + """A run-now by job ID targets a job outside the source; its job_name is only the caller's task key.""" + pipeline = Pipeline( + name="dag_main", + tasks=[RunJobActivity(name="trigger", task_key="trigger", job_name="trigger", existing_job_id="123")], + ) + + assert build_control_edges(pipeline) == [] + + def test_control_edges_unresolved_callee_is_recorded_not_dropped(): """An empty callee is kept with resolved=False rather than silently dropped.""" pipeline = Pipeline(name="parent", tasks=[_execute("call", "")]) From d64624cad0fd8dd16756370aadce905d3098d7c4 Mon Sep 17 00:00:00 2001 From: Matthew Moorcroft Date: Wed, 7 Oct 2026 15:43:36 +0100 Subject: [PATCH 11/25] no-mistakes(review): Keep bracket-quoted SQL names physical; reword generated_at doc note --- skills/flowx-discover/sources/adf.md | 4 +-- src/flowx/sources/adf/dataset_lineage.py | 8 +++--- tests/unit/test_adf_dataset_lineage.py | 33 +++++++++++++++++++++++- 3 files changed, 39 insertions(+), 6 deletions(-) diff --git a/skills/flowx-discover/sources/adf.md b/skills/flowx-discover/sources/adf.md index a188db40..f79205c8 100644 --- a/skills/flowx-discover/sources/adf.md +++ b/skills/flowx-discover/sources/adf.md @@ -66,8 +66,8 @@ Each activity's `task_key` equals its ADF activity name. `source_graphs_sha256` `metadata/source_graphs.json` the inventory was built from. The inventory deliberately has no top-level `generated_at` timestamp any more, so the same export -always produces the same `inventory.json` bytes and the hashes that bind enrich and route to it stay -stable. Use the file's modification time if you need to know when discover ran. +always produces byte-identical `inventory.json` and any hash taken over it stays stable. Use the +file's modification time if you need to know when discover ran. ## Step 4b — Review the complexity report diff --git a/src/flowx/sources/adf/dataset_lineage.py b/src/flowx/sources/adf/dataset_lineage.py index a0ce7806..eeadea6f 100644 --- a/src/flowx/sources/adf/dataset_lineage.py +++ b/src/flowx/sources/adf/dataset_lineage.py @@ -41,6 +41,7 @@ _ACCOUNT_NAME_RE = re.compile(r"AccountName=([A-Za-z0-9]+)", re.IGNORECASE) _DATASET_PARAM_RE = re.compile(r"^@dataset\(\)\.([A-Za-z_][A-Za-z0-9_]*)$") +_ARM_EXPRESSION_RE = re.compile(r"^\[\s*[A-Za-z_][A-Za-z0-9_]*\s*\(.*\]$", re.DOTALL) # Runtime references inside a path expression. Each is a value only knowable at # run time; for a *structural* signature we collapse them all to one slot token so @@ -501,13 +502,14 @@ def _is_physical(value: str) -> bool: DAB-ref placeholder (``{{`` ... ``}}``), a leftover ADF interpolation fragment (``@{``), a bare ADF expression (starts with ``@``), or an ARM template expression the export left unevaluated (``[parameters('x')]``, - ``[concat(...)]``). ARM writes a literal that starts with ``[`` as ``[[``. + ``[concat(...)]``). An ARM expression always opens with a function call, so + bracket-quoted SQL names such as ``[dbo].[Orders]`` stay physical, and so + does ARM's ``[[`` escape for a literal that starts with ``[``. """ stripped = value.strip() if "{{" in value or "@{" in value or stripped.startswith("@"): return False - is_arm_expression = stripped.startswith("[") and not stripped.startswith("[[") and stripped.endswith("]") - return not is_arm_expression + return not _ARM_EXPRESSION_RE.match(stripped) # --------------------------------------------------------------------------- # diff --git a/tests/unit/test_adf_dataset_lineage.py b/tests/unit/test_adf_dataset_lineage.py index 22923271..9037ca6f 100644 --- a/tests/unit/test_adf_dataset_lineage.py +++ b/tests/unit/test_adf_dataset_lineage.py @@ -487,7 +487,7 @@ def test_unevaluated_arm_expressions_never_become_identities() -> None: "ds_file": _adls_dataset( "ds_file", file_system="[concat('raw', parameters('env'))]", folder_path="in", linked_service="ls" ), - "ds_escaped": _adls_dataset("ds_escaped", file_system="[[literal", folder_path="in", linked_service="ls"), + "ds_escaped": _adls_dataset("ds_escaped", file_system="[[literal]", folder_path="in", linked_service="ls"), }, linked_services={ "ls_sql": AdfLinkedService( @@ -502,6 +502,37 @@ def test_unevaluated_arm_expressions_never_become_identities() -> None: assert resolve_dataset_identity(AdfDatasetReference(reference_name="ds_escaped"), definitions) is not None +def test_bracket_quoted_sql_names_keep_their_identity() -> None: + """Bracket-quoted SQL names are literal table names, not ARM expressions, so they keep their identity.""" + definitions = AdfDefinitions( + pipelines=[], + datasets={ + "ds_legacy": AdfDataset( + name="ds_legacy", + type="AzureSqlTable", + properties={ + "typeProperties": {"tableName": "[dbo].[Orders]"}, + "linkedServiceName": {"referenceName": "ls_sql"}, + }, + ), + "ds_spaced": _table_dataset("ds_spaced", schema="dbo", table="[Order Details]"), + }, + linked_services={ + "ls_sql": AdfLinkedService( + name="ls_sql", type="AzureSqlDatabase", properties={"typeProperties": {"server": "sql.example.net"}} + ), + }, + ) + assert ( + resolve_dataset_identity(AdfDatasetReference(reference_name="ds_legacy"), definitions) + == "sql.example.net/[dbo].[Orders]" + ) + assert ( + resolve_dataset_identity(AdfDatasetReference(reference_name="ds_spaced"), definitions) + == "sql.example.net/dbo.[Order Details]" + ) + + def test_parameterised_file_location_bound_to_literals_resolves_to_its_path() -> None: """``@dataset().x`` location parts resolve against the call-site bindings, as table names do.""" definitions = AdfDefinitions( From adf84b57c0dee6b9dcd6be99e1469919611f2d74 Mon Sep 17 00:00:00 2001 From: Matthew Moorcroft Date: Wed, 7 Oct 2026 16:06:45 +0100 Subject: [PATCH 12/25] no-mistakes(review): Drop identities for storeSettings-overridden reads and ARM storage URLs --- src/flowx/sources/adf/dataset_lineage.py | 54 +++++++++++--- tests/unit/test_adf_dataset_lineage.py | 90 ++++++++++++++++++++++++ 2 files changed, 135 insertions(+), 9 deletions(-) diff --git a/src/flowx/sources/adf/dataset_lineage.py b/src/flowx/sources/adf/dataset_lineage.py index eeadea6f..d80d40e0 100644 --- a/src/flowx/sources/adf/dataset_lineage.py +++ b/src/flowx/sources/adf/dataset_lineage.py @@ -42,6 +42,7 @@ _ACCOUNT_NAME_RE = re.compile(r"AccountName=([A-Za-z0-9]+)", re.IGNORECASE) _DATASET_PARAM_RE = re.compile(r"^@dataset\(\)\.([A-Za-z_][A-Za-z0-9_]*)$") _ARM_EXPRESSION_RE = re.compile(r"^\[\s*[A-Za-z_][A-Za-z0-9_]*\s*\(.*\]$", re.DOTALL) +_RUN_TIME_PATH_OVERRIDES = ("wildcardFolderPath", "wildcardFileName", "fileListPath", "prefix") # Runtime references inside a path expression. Each is a value only knowable at # run time; for a *structural* signature we collapse them all to one slot token so @@ -75,6 +76,11 @@ def activity_data_assets( included. A dataset named in both an activity-level slot and ``typeProperties`` is counted once per side so an activity does not emit two identical assets. + When the activity's ``storeSettings`` override the read path at run time (a + wildcard folder or file name, a file list, or a prefix), the dataset's own + location is not what the activity reads, so its reads get no identity and fall + back to the structural signature. + Args: activity: The ADF activity to resolve. definitions: All loaded ADF definitions (datasets + linked services), @@ -87,8 +93,9 @@ def activity_data_assets( ``(data_reads, data_writes)`` as lists of :class:`DataAsset`. """ resolution_context = context if context is not None else TranslationContext() + read_location_overridden = _read_location_overridden(activity) reads = [ - _dataset_ref_to_asset(reference, definitions, resolution_context) + _dataset_ref_to_asset(reference, definitions, resolution_context, location_overridden=read_location_overridden) for reference in _activity_dataset_refs(activity, produced=False) ] writes = [ @@ -98,10 +105,30 @@ def activity_data_assets( return reads, writes +def _read_location_overridden(activity: AdfActivity) -> bool: + """Say whether the activity's ``storeSettings`` replace the read dataset's path at run time. + + Copy and Lookup carry them on ``typeProperties.source``; Delete and GetMetadata + carry them directly on ``typeProperties``. + """ + type_properties = activity.type_properties or {} + source = type_properties.get("source") + store_settings_candidates = ( + source.get("storeSettings") if isinstance(source, dict) else None, + type_properties.get("storeSettings"), + ) + return any( + isinstance(store_settings, dict) and any(store_settings.get(key) for key in _RUN_TIME_PATH_OVERRIDES) + for store_settings in store_settings_candidates + ) + + def _dataset_ref_to_asset( dataset_ref: AdfDatasetReference, definitions: AdfDefinitions, context: TranslationContext, + *, + location_overridden: bool = False, ) -> DataAsset: """Turn one dataset reference into a two-tier :class:`DataAsset`. @@ -112,9 +139,10 @@ def _dataset_ref_to_asset( path-anchored signature is available the signature is left empty. An empty signature is falsy, so :func:`~flowx.lineage._match_assets` cannot use it as a join key -- the asset is still captured as a read / write for reporting, it just - cannot manufacture a signature-tier edge. + cannot manufacture a signature-tier edge. When *location_overridden* is set the + dataset's location is not what the activity touches, so no identity is resolved. """ - identity = resolve_dataset_identity(dataset_ref, definitions, context) + identity = None if location_overridden else resolve_dataset_identity(dataset_ref, definitions, context) if identity is not None: # Mirror the identity into the signature so the weak tier never joins a # resolved asset to an unresolved one that merely shares a physical value. @@ -460,13 +488,20 @@ def _resolved_part(raw: Any) -> str | None: def _resolve_storage_account(linked_service: Any) -> str | None: - """Pull a storage account name out of a linked service, if present.""" + """Pull a storage account name out of a linked service, if present. + + The whole ``url``, ``sasUri`` or connection string must be physical before the + account is cut out of it: cutting first would drop the closing ``]`` of an ARM + expression or the tail of an ``@{...}`` and leave a fragment that looks literal. + """ if linked_service is None: return None type_props = linked_service.properties.get("typeProperties") or linked_service.properties url = type_props.get("url") or "" if isinstance(url, str) and url: + if not _is_physical(url): + return None host = url.replace("https://", "").split("/", 1)[0] host_no_port = host.split(":", 1)[0] if "." in host_no_port: @@ -474,21 +509,22 @@ def _resolve_storage_account(linked_service: Any) -> str | None: sas_uri = type_props.get("sasUri") or "" if isinstance(sas_uri, str) and sas_uri: + if not _is_physical(sas_uri): + return None host = sas_uri.split("?", 1)[0].replace("https://", "").split("/", 1)[0] if "." in host: return host.split(".", 1)[0] # Plaintext connection string (rare in az exports -- usually masked). connection_string = type_props.get("connectionString") + if isinstance(connection_string, dict): + connection_string = connection_string.get("value", "") if isinstance(connection_string, str): + if not _is_physical(connection_string): + return None match = _ACCOUNT_NAME_RE.search(connection_string) if match: return match.group(1) - if isinstance(connection_string, dict): - value = connection_string.get("value", "") - match = _ACCOUNT_NAME_RE.search(value) - if match: - return match.group(1) # AWS -- bucket name lives on the dataset, account is implicit; nothing useful # to return at the linked-service level for S3 / GCS. diff --git a/tests/unit/test_adf_dataset_lineage.py b/tests/unit/test_adf_dataset_lineage.py index 9037ca6f..2c2704af 100644 --- a/tests/unit/test_adf_dataset_lineage.py +++ b/tests/unit/test_adf_dataset_lineage.py @@ -9,6 +9,7 @@ from __future__ import annotations +from flowx.lineage import data_edges_from_endpoints from flowx.models.adf_ast import ( AdfActivity, AdfDataset, @@ -533,6 +534,95 @@ def test_bracket_quoted_sql_names_keep_their_identity() -> None: ) +def test_arm_expression_storage_url_has_no_path_identity() -> None: + """An unevaluated ARM ``url`` is checked whole, so a cut-off fragment never stands in for the account.""" + arm_lake = AdfLinkedService( + name="ls_arm_lake", + type="AzureBlobFS", + properties={ + "typeProperties": {"url": "[concat('https://', parameters('storageAccountName'), '.dfs.core.windows.net')]"} + }, + ) + definitions = AdfDefinitions( + pipelines=[], + datasets={ + "ds_orders": _adls_dataset( + "ds_orders", file_system="raw", folder_path="in", linked_service="ls_arm_lake", file_name="orders.csv" + ) + }, + linked_services={"ls_arm_lake": arm_lake}, + ) + assert resolve_dataset_identity(AdfDatasetReference(reference_name="ds_orders"), definitions) is None + + +def _lake_definitions() -> AdfDefinitions: + return AdfDefinitions( + pipelines=[], + datasets={"ds_lake": _adls_dataset("ds_lake", file_system="data", folder_path="landing", linked_service="ls")}, + linked_services={"ls": _adls_linked_service("ls", account="acct")}, + ) + + +def test_store_settings_path_override_read_gets_no_identity_edge() -> None: + """A Copy that reads through a wildcard override does not read the dataset's folder, so no identity edge forms.""" + definitions = _lake_definitions() + ingest = AdfActivity( + name="Ingest", + type="Copy", + outputs=[AdfDatasetReference(reference_name="ds_lake")], + type_properties={"sink": {"type": "DelimitedTextSink"}}, + ) + publish = AdfActivity( + name="Publish", + type="Copy", + inputs=[AdfDatasetReference(reference_name="ds_lake")], + type_properties={ + "source": { + "type": "DelimitedTextSource", + "storeSettings": {"wildcardFolderPath": "archive/2023", "wildcardFileName": "*.csv"}, + } + }, + ) + _, ingest_writes = activity_data_assets(ingest, definitions) + publish_reads, _ = activity_data_assets(publish, definitions) + + assert ingest_writes[0].identity == "abfss://data@acct.dfs.core.windows.net/landing" + assert publish_reads[0].identity is None + edges = data_edges_from_endpoints( + [("Ingest", asset) for asset in ingest_writes], [("Publish", asset) for asset in publish_reads] + ) + assert edges == [] + + +def test_store_settings_path_override_covers_lookup_and_dataset_activities() -> None: + """Lookup overrides on its ``source``; Delete and GetMetadata override directly on ``typeProperties``.""" + definitions = _lake_definitions() + lookup = AdfActivity( + name="Lookup", + type="Lookup", + type_properties={ + "source": {"type": "DelimitedTextSource", "storeSettings": {"prefix": "orders_"}}, + "dataset": {"referenceName": "ds_lake"}, + }, + ) + get_metadata = AdfActivity( + name="GetMetadata", + type="GetMetadata", + type_properties={"dataset": {"referenceName": "ds_lake"}, "storeSettings": {"fileListPath": "lists/today.txt"}}, + ) + recursive_delete = AdfActivity( + name="Delete", + type="Delete", + type_properties={"dataset": {"referenceName": "ds_lake"}, "storeSettings": {"recursive": True}}, + ) + + assert [asset.identity for asset in activity_data_assets(lookup, definitions)[0]] == [None] + assert [asset.identity for asset in activity_data_assets(get_metadata, definitions)[0]] == [None] + assert [asset.identity for asset in activity_data_assets(recursive_delete, definitions)[0]] == [ + "abfss://data@acct.dfs.core.windows.net/landing" + ] + + def test_parameterised_file_location_bound_to_literals_resolves_to_its_path() -> None: """``@dataset().x`` location parts resolve against the call-site bindings, as table names do.""" definitions = AdfDefinitions( From a911dc7fb2253a20e9c3ee46e8479d22d8eb17f5 Mon Sep 17 00:00:00 2001 From: Matthew Moorcroft Date: Wed, 7 Oct 2026 16:19:08 +0100 Subject: [PATCH 13/25] no-mistakes(review): Drop identities for query, procedure and path-overridden datasets --- src/flowx/sources/adf/dataset_lineage.py | 44 +++++++++++------ tests/unit/test_adf_dataset_lineage.py | 62 ++++++++++++++++++++++++ 2 files changed, 91 insertions(+), 15 deletions(-) diff --git a/src/flowx/sources/adf/dataset_lineage.py b/src/flowx/sources/adf/dataset_lineage.py index d80d40e0..3b463724 100644 --- a/src/flowx/sources/adf/dataset_lineage.py +++ b/src/flowx/sources/adf/dataset_lineage.py @@ -43,6 +43,8 @@ _DATASET_PARAM_RE = re.compile(r"^@dataset\(\)\.([A-Za-z_][A-Za-z0-9_]*)$") _ARM_EXPRESSION_RE = re.compile(r"^\[\s*[A-Za-z_][A-Za-z0-9_]*\s*\(.*\]$", re.DOTALL) _RUN_TIME_PATH_OVERRIDES = ("wildcardFolderPath", "wildcardFileName", "fileListPath", "prefix") +_SOURCE_QUERY_OVERRIDES = ("sqlReaderQuery", "query", "sqlReaderStoredProcedureName") +_SINK_PROCEDURE_OVERRIDES = ("sqlWriterStoredProcedureName",) # Runtime references inside a path expression. Each is a value only knowable at # run time; for a *structural* signature we collapse them all to one slot token so @@ -76,10 +78,12 @@ def activity_data_assets( included. A dataset named in both an activity-level slot and ``typeProperties`` is counted once per side so an activity does not emit two identical assets. - When the activity's ``storeSettings`` override the read path at run time (a - wildcard folder or file name, a file list, or a prefix), the dataset's own - location is not what the activity reads, so its reads get no identity and fall - back to the structural signature. + When the activity overrides the dataset's physical source or target at run time, + the dataset's own location is not what the activity touches, so that side gets + no identity and falls back to the structural signature. Reads are overridden by + a source query or stored procedure, or by ``storeSettings`` that give a wildcard + folder or file name, a file list, or a prefix; writes by a sink stored procedure, + which decides the table itself. Args: activity: The ADF activity to resolve. @@ -93,42 +97,52 @@ def activity_data_assets( ``(data_reads, data_writes)`` as lists of :class:`DataAsset`. """ resolution_context = context if context is not None else TranslationContext() - read_location_overridden = _read_location_overridden(activity) + reads_overridden = _location_overridden(activity, produced=False) reads = [ - _dataset_ref_to_asset(reference, definitions, resolution_context, location_overridden=read_location_overridden) + _dataset_ref_to_asset(reference, definitions, resolution_context, location_overridden=reads_overridden) for reference in _activity_dataset_refs(activity, produced=False) ] + writes_overridden = _location_overridden(activity, produced=True) writes = [ - _dataset_ref_to_asset(reference, definitions, resolution_context) + _dataset_ref_to_asset(reference, definitions, resolution_context, location_overridden=writes_overridden) for reference in _activity_dataset_refs(activity, produced=True) ] return reads, writes -def _read_location_overridden(activity: AdfActivity) -> bool: - """Say whether the activity's ``storeSettings`` replace the read dataset's path at run time. +def _location_overridden(activity: AdfActivity, *, produced: bool) -> bool: + """Say whether the activity replaces its datasets' physical target (produced) or source (not) at run time. - Copy and Lookup carry them on ``typeProperties.source``; Delete and GetMetadata - carry them directly on ``typeProperties``. + A Copy sink that names a stored procedure writes wherever the procedure decides. + A Copy or Lookup ``source`` that names a query or stored procedure reads what it + returns. ``storeSettings`` with a wildcard, file list or prefix replace the read + path: Copy and Lookup carry them on ``source``, Delete and GetMetadata directly + on ``typeProperties``. """ type_properties = activity.type_properties or {} + if produced: + return _names_any(type_properties.get("sink"), _SINK_PROCEDURE_OVERRIDES) source = type_properties.get("source") store_settings_candidates = ( source.get("storeSettings") if isinstance(source, dict) else None, type_properties.get("storeSettings"), ) - return any( - isinstance(store_settings, dict) and any(store_settings.get(key) for key in _RUN_TIME_PATH_OVERRIDES) - for store_settings in store_settings_candidates + return _names_any(source, _SOURCE_QUERY_OVERRIDES) or any( + _names_any(store_settings, _RUN_TIME_PATH_OVERRIDES) for store_settings in store_settings_candidates ) +def _names_any(settings: object, keys: tuple[str, ...]) -> bool: + """``True`` when *settings* is a dict that gives a non-empty value for any of *keys*.""" + return isinstance(settings, dict) and any(settings.get(key) for key in keys) + + def _dataset_ref_to_asset( dataset_ref: AdfDatasetReference, definitions: AdfDefinitions, context: TranslationContext, *, - location_overridden: bool = False, + location_overridden: bool, ) -> DataAsset: """Turn one dataset reference into a two-tier :class:`DataAsset`. diff --git a/tests/unit/test_adf_dataset_lineage.py b/tests/unit/test_adf_dataset_lineage.py index 2c2704af..c0ca2378 100644 --- a/tests/unit/test_adf_dataset_lineage.py +++ b/tests/unit/test_adf_dataset_lineage.py @@ -9,6 +9,8 @@ from __future__ import annotations +from pathlib import Path + from flowx.lineage import data_edges_from_endpoints from flowx.models.adf_ast import ( AdfActivity, @@ -18,6 +20,9 @@ AdfLinkedService, ) from flowx.sources.adf.dataset_lineage import activity_data_assets, resolve_dataset_identity +from flowx.sources.adf.loader import load_adf_definitions + +FIXTURES_DIR = Path(__file__).parent.parent / "resources" / "json" def _definitions(**datasets: AdfDataset) -> AdfDefinitions: @@ -623,6 +628,63 @@ def test_store_settings_path_override_covers_lookup_and_dataset_activities() -> ] +def test_source_query_read_gets_no_identity_edge() -> None: + """A Lookup that runs its own query reads what the query returns, not the dataset's table.""" + definitions = load_adf_definitions(FIXTURES_DIR) + load_orders = AdfActivity( + name="LoadOrders", + type="Copy", + outputs=[AdfDatasetReference(reference_name="ds_azure_sql_orders")], + type_properties={"sink": {"type": "AzureSqlSink"}}, + ) + get_watermark = AdfActivity( + name="GetWatermark", + type="Lookup", + type_properties={ + "source": {"type": "AzureSqlSource", "sqlReaderQuery": "SELECT MAX(ts) AS wm FROM dbo.watermarks"}, + "dataset": {"referenceName": "ds_azure_sql_orders"}, + }, + ) + _, load_orders_writes = activity_data_assets(load_orders, definitions) + get_watermark_reads, _ = activity_data_assets(get_watermark, definitions) + + assert load_orders_writes[0].identity == "ls_azure_sql/dbo.orders" + assert get_watermark_reads[0].identity is None + edges = data_edges_from_endpoints( + [("LoadOrders", asset) for asset in load_orders_writes], + [("GetWatermark", asset) for asset in get_watermark_reads], + ) + assert edges == [] + + +def test_stored_procedure_source_and_sink_get_no_identity() -> None: + """A stored procedure decides what is read or written, so neither side keeps the dataset's table identity.""" + definitions = _definitions(ds_orders=_table_dataset("ds_orders", schema="dbo", table="Orders")) + lookup = AdfActivity( + name="LookupStoredProc", + type="Lookup", + type_properties={ + "source": {"type": "SqlSource", "sqlReaderStoredProcedureName": "dbo.usp_get_orders"}, + "dataset": {"referenceName": "ds_orders"}, + }, + ) + copy = AdfActivity( + name="UpsertOrders", + type="Copy", + inputs=[AdfDatasetReference(reference_name="ds_orders")], + outputs=[AdfDatasetReference(reference_name="ds_orders")], + type_properties={ + "source": {"type": "SqlSource"}, + "sink": {"type": "SqlSink", "sqlWriterStoredProcedureName": "dbo.usp_upsert_orders"}, + }, + ) + copy_reads, copy_writes = activity_data_assets(copy, definitions) + + assert [asset.identity for asset in activity_data_assets(lookup, definitions)[0]] == [None] + assert [asset.identity for asset in copy_writes] == [None] + assert [asset.identity for asset in copy_reads] == ["ls_sql/dbo.Orders"] + + def test_parameterised_file_location_bound_to_literals_resolves_to_its_path() -> None: """``@dataset().x`` location parts resolve against the call-site bindings, as table names do.""" definitions = AdfDefinitions( From a4bae0e09607a5562a2de661e401792c9bf03131 Mon Sep 17 00:00:00 2001 From: Matthew Moorcroft Date: Wed, 7 Oct 2026 16:29:38 +0100 Subject: [PATCH 14/25] no-mistakes(review): Match source query and procedure overrides by key kind --- src/flowx/sources/adf/dataset_lineage.py | 31 ++++++++----- tests/unit/test_adf_dataset_lineage.py | 57 ++++++++++++++++++++++++ 2 files changed, 76 insertions(+), 12 deletions(-) diff --git a/src/flowx/sources/adf/dataset_lineage.py b/src/flowx/sources/adf/dataset_lineage.py index 3b463724..6ada8f3f 100644 --- a/src/flowx/sources/adf/dataset_lineage.py +++ b/src/flowx/sources/adf/dataset_lineage.py @@ -32,7 +32,7 @@ import json import re -from collections.abc import Iterator +from collections.abc import Callable, Iterator from typing import Any from flowx.models.adf_ast import AdfActivity, AdfDatasetReference, AdfDefinitions @@ -43,8 +43,6 @@ _DATASET_PARAM_RE = re.compile(r"^@dataset\(\)\.([A-Za-z_][A-Za-z0-9_]*)$") _ARM_EXPRESSION_RE = re.compile(r"^\[\s*[A-Za-z_][A-Za-z0-9_]*\s*\(.*\]$", re.DOTALL) _RUN_TIME_PATH_OVERRIDES = ("wildcardFolderPath", "wildcardFileName", "fileListPath", "prefix") -_SOURCE_QUERY_OVERRIDES = ("sqlReaderQuery", "query", "sqlReaderStoredProcedureName") -_SINK_PROCEDURE_OVERRIDES = ("sqlWriterStoredProcedureName",) # Runtime references inside a path expression. Each is a value only knowable at # run time; for a *structural* signature we collapse them all to one slot token so @@ -115,26 +113,35 @@ def _location_overridden(activity: AdfActivity, *, produced: bool) -> bool: A Copy sink that names a stored procedure writes wherever the procedure decides. A Copy or Lookup ``source`` that names a query or stored procedure reads what it - returns. ``storeSettings`` with a wildcard, file list or prefix replace the read - path: Copy and Lookup carry them on ``source``, Delete and GetMetadata directly - on ``typeProperties``. + returns. Connectors name these keys differently (``sqlReaderQuery``, + ``oracleReaderQuery``, ``sqlReaderStoredProcedureName``, ...), so they are + matched by kind: ``query`` or any key ending in ``Query`` or + ``StoredProcedureName``. ``storeSettings`` with a wildcard, file list or prefix + replace the read path: Copy and Lookup carry them on ``source``, Delete and + GetMetadata directly on ``typeProperties``. """ type_properties = activity.type_properties or {} if produced: - return _names_any(type_properties.get("sink"), _SINK_PROCEDURE_OVERRIDES) + return _names_any(type_properties.get("sink"), lambda key: key.endswith("StoredProcedureName")) source = type_properties.get("source") store_settings_candidates = ( source.get("storeSettings") if isinstance(source, dict) else None, type_properties.get("storeSettings"), ) - return _names_any(source, _SOURCE_QUERY_OVERRIDES) or any( - _names_any(store_settings, _RUN_TIME_PATH_OVERRIDES) for store_settings in store_settings_candidates + return _names_any(source, _is_source_query_key) or any( + _names_any(store_settings, lambda key: key in _RUN_TIME_PATH_OVERRIDES) + for store_settings in store_settings_candidates ) -def _names_any(settings: object, keys: tuple[str, ...]) -> bool: - """``True`` when *settings* is a dict that gives a non-empty value for any of *keys*.""" - return isinstance(settings, dict) and any(settings.get(key) for key in keys) +def _is_source_query_key(key: str) -> bool: + """``True`` for a ``source`` key that names a query or stored procedure to read through.""" + return key == "query" or key.endswith("Query") or key.endswith("StoredProcedureName") + + +def _names_any(settings: object, is_override_key: Callable[[str], bool]) -> bool: + """``True`` when *settings* is a dict with a non-empty value under a key *is_override_key* accepts.""" + return isinstance(settings, dict) and any(value and is_override_key(key) for key, value in settings.items()) def _dataset_ref_to_asset( diff --git a/tests/unit/test_adf_dataset_lineage.py b/tests/unit/test_adf_dataset_lineage.py index c0ca2378..111d8735 100644 --- a/tests/unit/test_adf_dataset_lineage.py +++ b/tests/unit/test_adf_dataset_lineage.py @@ -685,6 +685,63 @@ def test_stored_procedure_source_and_sink_get_no_identity() -> None: assert [asset.identity for asset in copy_reads] == ["ls_sql/dbo.Orders"] +def test_connector_specific_query_read_gets_no_identity_edge() -> None: + """An Oracle ``oracleReaderQuery`` overrides the read like ``sqlReaderQuery``; an empty one does not.""" + definitions = AdfDefinitions( + pipelines=[], + datasets={ + "ds_oracle_control": AdfDataset( + name="ds_oracle_control", + type="OracleTable", + properties={ + "typeProperties": {"schema": "HR", "table": "ETL_CONTROL"}, + "linkedServiceName": {"referenceName": "ls_oracle"}, + }, + ) + }, + linked_services={ + "ls_oracle": AdfLinkedService( + name="ls_oracle", + type="Oracle", + properties={"typeProperties": {"connectionString": "host=ora.example.net;port=1521;serviceName=HR"}}, + ) + }, + ) + stage_control = AdfActivity( + name="StageControl", + type="Copy", + outputs=[AdfDatasetReference(reference_name="ds_oracle_control")], + type_properties={"sink": {"type": "OracleSink"}}, + ) + + def _lookup(name: str, query: str) -> AdfActivity: + return AdfActivity( + name=name, + type="Lookup", + type_properties={ + "source": {"type": "OracleSource", "oracleReaderQuery": query}, + "dataset": {"referenceName": "ds_oracle_control"}, + }, + ) + + _, stage_control_writes = activity_data_assets(stage_control, definitions) + get_watermark_reads, _ = activity_data_assets( + _lookup("GetWatermark", "SELECT MAX(loaded_at) FROM HR.ETL_AUDIT"), definitions + ) + read_control_reads, _ = activity_data_assets(_lookup("ReadControl", ""), definitions) + + assert stage_control_writes[0].identity == "ls_oracle/HR.ETL_CONTROL" + assert get_watermark_reads[0].identity is None + assert ( + data_edges_from_endpoints( + [("StageControl", asset) for asset in stage_control_writes], + [("GetWatermark", asset) for asset in get_watermark_reads], + ) + == [] + ) + assert read_control_reads[0].identity == "ls_oracle/HR.ETL_CONTROL" + + def test_parameterised_file_location_bound_to_literals_resolves_to_its_path() -> None: """``@dataset().x`` location parts resolve against the call-site bindings, as table names do.""" definitions = AdfDefinitions( From 79f70e3201ee547852bd21f4584a3a18f45655ff Mon Sep 17 00:00:00 2001 From: Matthew Moorcroft Date: Wed, 7 Oct 2026 16:52:49 +0100 Subject: [PATCH 15/25] no-mistakes(document): Document run-now skip in build_control_edges docstring --- src/flowx/lineage.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/src/flowx/lineage.py b/src/flowx/lineage.py index c8c686e2..52183300 100644 --- a/src/flowx/lineage.py +++ b/src/flowx/lineage.py @@ -126,7 +126,8 @@ def build_control_edges(pipeline: Pipeline) -> list[ControlEdge]: (fan-out is preserved: each call site is its own edge). Edges whose callee equals the caller are dropped (no self-edges), and identical edges are collapsed (no duplicates). An unresolved callee is recorded with - ``resolved=False`` rather than dropped. + ``resolved=False`` rather than dropped. A ``RunJobActivity`` that runs an + existing job by ID emits no edge. Args: pipeline: The translated pipeline IR. From efd820864bdbe7d7054f8130dba11ba1613ed22e Mon Sep 17 00:00:00 2001 From: Matthew Moorcroft Date: Wed, 7 Oct 2026 20:31:42 +0100 Subject: [PATCH 16/25] Drop signatures on overridden reads, keep empty Switch cases, carry motif hints A read whose activity overrides its dataset location at run time (storeSettings paths, a SQL query or a stored procedure) now gets neither an identity nor a signature, so it cannot join a writer of the same dataset binding on the weak tier either. A Switch case declared with no activities keeps its label as an empty branch in the saved graph, recovered from typeProperties in declaration order. IR lineage motif annotations now carry the detector's source_type_hint. Each fix has a regression test that fails on 79f70e3. Co-authored-by: Isaac --- src/flowx/lineage.py | 1 + src/flowx/sources/adf/dataset_lineage.py | 7 ++-- src/flowx/sources/adf/discovery_mapping.py | 13 +++++-- tests/unit/test_adf_dataset_lineage.py | 42 ++++++++++++++++++++++ tests/unit/test_adf_discovery_mapping.py | 24 +++++++++++++ tests/unit/test_lineage_substrate.py | 22 ++++++++++++ 6 files changed, 105 insertions(+), 4 deletions(-) diff --git a/src/flowx/lineage.py b/src/flowx/lineage.py index 52183300..d88dea8c 100644 --- a/src/flowx/lineage.py +++ b/src/flowx/lineage.py @@ -278,6 +278,7 @@ def build_motif_annotations(pipeline: Pipeline) -> list[MotifAnnotation]: display_name=activity.display_name, databricks_replacement=activity.databricks_replacement, notes=list(activity.confidence_notes), + source_type_hint=getattr(activity, "source_type_hint", None), ) ) return annotations diff --git a/src/flowx/sources/adf/dataset_lineage.py b/src/flowx/sources/adf/dataset_lineage.py index 6ada8f3f..94d31887 100644 --- a/src/flowx/sources/adf/dataset_lineage.py +++ b/src/flowx/sources/adf/dataset_lineage.py @@ -161,9 +161,12 @@ def _dataset_ref_to_asset( signature is falsy, so :func:`~flowx.lineage._match_assets` cannot use it as a join key -- the asset is still captured as a read / write for reporting, it just cannot manufacture a signature-tier edge. When *location_overridden* is set the - dataset's location is not what the activity touches, so no identity is resolved. + dataset's location is not what the activity touches, so neither an identity nor a + signature is derived from it. """ - identity = None if location_overridden else resolve_dataset_identity(dataset_ref, definitions, context) + if location_overridden: + return DataAsset(signature="", identity=None, asset_type=_asset_type(dataset_ref, definitions)) + identity = resolve_dataset_identity(dataset_ref, definitions, context) if identity is not None: # Mirror the identity into the signature so the weak tier never joins a # resolved asset to an unresolved one that merely shares a physical value. diff --git a/src/flowx/sources/adf/discovery_mapping.py b/src/flowx/sources/adf/discovery_mapping.py index d96fb869..31e41d9b 100644 --- a/src/flowx/sources/adf/discovery_mapping.py +++ b/src/flowx/sources/adf/discovery_mapping.py @@ -341,8 +341,17 @@ def _control_flow_branches(activity: AdfActivity, definitions: AdfDefinitions) - return {"body": [_activity_to_node(child, definitions) for child in (activity.activities or [])]} if activity.type == "Switch": branches: dict[str, list[SourceNode]] = {} - for case_value, case_activities in (activity.switch_cases or {}).items(): - branches[case_value] = [_activity_to_node(child, definitions) for child in case_activities] + parsed_cases = activity.switch_cases or {} + # The parser drops a case whose activities list is empty, so the declared case labels come + # from the raw typeProperties; that keeps every named branch, empty ones included. + declared_cases = (activity.type_properties or {}).get("cases") + declared_values = [ + str(case["value"]) + for case in declared_cases or [] + if isinstance(case, dict) and case.get("value") is not None + ] + for case_value in declared_values + [value for value in parsed_cases if value not in declared_values]: + branches[case_value] = [_activity_to_node(child, definitions) for child in parsed_cases.get(case_value, [])] branches["default"] = [ _activity_to_node(child, definitions) for child in (activity.switch_default_activities or []) ] diff --git a/tests/unit/test_adf_dataset_lineage.py b/tests/unit/test_adf_dataset_lineage.py index 111d8735..69a3b6ef 100644 --- a/tests/unit/test_adf_dataset_lineage.py +++ b/tests/unit/test_adf_dataset_lineage.py @@ -766,3 +766,45 @@ def test_parameterised_file_location_bound_to_literals_resolves_to_its_path() -> ) # An unbound file-name parameter is unknown, so the folder alone is never used as the identity. assert resolve_dataset_identity(unbound, definitions) is None + + +def test_overridden_read_carries_no_signature_so_it_cannot_join_on_the_weak_tier() -> None: + """With the writer's identity unresolved, a shared dataset binding must not join a wildcard reader by signature.""" + definitions = AdfDefinitions( + pipelines=[], + datasets={ + "ds_param": _adls_dataset( + "ds_param", + file_system="data", + folder_path="@dataset().folderPath", + linked_service="ls_secret", + ) + }, + linked_services={ + "ls_secret": AdfLinkedService(name="ls_secret", type="AzureBlobFS", properties={"typeProperties": {}}) + }, + ) + binding = {"folderPath": {"value": "@concat('landing/', pipeline().parameters.run)", "type": "Expression"}} + ingest = AdfActivity( + name="Ingest", + type="Copy", + outputs=[AdfDatasetReference(reference_name="ds_param", parameters=binding)], + type_properties={"sink": {"type": "DelimitedTextSink"}}, + ) + publish = AdfActivity( + name="Publish", + type="Copy", + inputs=[AdfDatasetReference(reference_name="ds_param", parameters=binding)], + type_properties={"source": {"type": "DelimitedTextSource", "storeSettings": {"wildcardFileName": "*.csv"}}}, + ) + _, ingest_writes = activity_data_assets(ingest, definitions) + publish_reads, _ = activity_data_assets(publish, definitions) + + assert ingest_writes[0].identity is None + assert ingest_writes[0].signature # the writer still has a path-anchored signature to join on + assert publish_reads[0].identity is None + assert publish_reads[0].signature == "" + edges = data_edges_from_endpoints( + [("Ingest", asset) for asset in ingest_writes], [("Publish", asset) for asset in publish_reads] + ) + assert edges == [] diff --git a/tests/unit/test_adf_discovery_mapping.py b/tests/unit/test_adf_discovery_mapping.py index a8037f81..5c532eb4 100644 --- a/tests/unit/test_adf_discovery_mapping.py +++ b/tests/unit/test_adf_discovery_mapping.py @@ -413,6 +413,30 @@ def test_one_sided_if_keeps_empty_false_branch() -> None: assert node.branches["false"] == [] +def test_switch_keeps_named_cases_that_have_no_activities() -> None: + """A declared case with ``activities: []`` keeps its label as an empty branch, in declaration order.""" + activities = [ + { + "name": "Route", + "type": "Switch", + "typeProperties": { + "cases": [ + {"value": "gold", "activities": [{"name": "G", "type": "Copy"}]}, + {"value": "skip", "activities": []}, + ], + "defaultActivities": [{"name": "D", "type": "Wait"}], + }, + } + ] + graph = _pipeline(activities) + + container = graph.tasks[0] + assert isinstance(container, ContainerNode) + assert list(container.branches.keys()) == ["gold", "skip", "default"] + assert container.branches["skip"] == [] + assert container.branches["gold"][0].name == "G" + + def test_empty_switch_stays_a_container_with_default_branch() -> None: """A Switch with no cases still maps to a ContainerNode with an empty default.""" graph = _pipeline([{"name": "Route", "type": "Switch", "typeProperties": {}}]) diff --git a/tests/unit/test_lineage_substrate.py b/tests/unit/test_lineage_substrate.py index b2a9c16e..b32a7372 100644 --- a/tests/unit/test_lineage_substrate.py +++ b/tests/unit/test_lineage_substrate.py @@ -339,6 +339,28 @@ def test_build_motif_annotations_groups_members_by_tag(): assert annotations[0].databricks_replacement == "auto_loader" +def test_build_motif_annotations_carries_the_source_type_hint(): + """The detector's source classification survives into the IR lineage annotation.""" + pipeline = Pipeline( + name="p", + tasks=[ + MotifActivity( + name="motif", + task_key="motif_bulk_copy", + motif_id="metadata_driven_bulk_copy", + display_name="Bulk copy", + databricks_replacement="lakeflow_connect", + matched_activity_names=["Lookup", "ForEach"], + source_type_hint="database", + ), + ], + ) + + (annotation,) = build_motif_annotations(pipeline) + + assert annotation.source_type_hint == "database" + + def test_with_lineage_is_pure(): """with_lineage returns a new pipeline and never mutates the input.""" pipeline = Pipeline(name="p", tasks=[_notebook("n")]) From bb6763e66bff5d9caf68402620d756adabe96f1d Mon Sep 17 00:00:00 2001 From: Matthew Moorcroft Date: Wed, 7 Oct 2026 21:14:39 +0100 Subject: [PATCH 17/25] no-mistakes(review): Key Switch cases like parser, fix override docstrings --- src/flowx/lineage.py | 2 +- src/flowx/sources/adf/dataset_lineage.py | 6 +++++- src/flowx/sources/adf/discovery_mapping.py | 7 +------ tests/unit/test_adf_discovery_mapping.py | 23 ++++++++++++++++++++++ 4 files changed, 30 insertions(+), 8 deletions(-) diff --git a/src/flowx/lineage.py b/src/flowx/lineage.py index d88dea8c..f89f4d54 100644 --- a/src/flowx/lineage.py +++ b/src/flowx/lineage.py @@ -278,7 +278,7 @@ def build_motif_annotations(pipeline: Pipeline) -> list[MotifAnnotation]: display_name=activity.display_name, databricks_replacement=activity.databricks_replacement, notes=list(activity.confidence_notes), - source_type_hint=getattr(activity, "source_type_hint", None), + source_type_hint=activity.source_type_hint, ) ) return annotations diff --git a/src/flowx/sources/adf/dataset_lineage.py b/src/flowx/sources/adf/dataset_lineage.py index 94d31887..13b3d329 100644 --- a/src/flowx/sources/adf/dataset_lineage.py +++ b/src/flowx/sources/adf/dataset_lineage.py @@ -23,6 +23,10 @@ references that merely share a name must not join (#36's explicit rule), so when neither a physical identity nor a path-anchored signature is available the signature is left empty and the asset cannot participate in signature matching. + The signature is also left empty when the activity overrides the dataset's + location at run time (a source query, stored procedure or ``storeSettings`` path + override on a read; a sink stored procedure on a write), because the dataset's + path is then not what the activity touches. Only literal, provable values ever become an ``identity`` -- the resolver returns ``None`` rather than guessing, which is what stopped #36's spurious edges. @@ -78,7 +82,7 @@ def activity_data_assets( When the activity overrides the dataset's physical source or target at run time, the dataset's own location is not what the activity touches, so that side gets - no identity and falls back to the structural signature. Reads are overridden by + neither an identity nor a signature and cannot join on either tier. Reads are overridden by a source query or stored procedure, or by ``storeSettings`` that give a wildcard folder or file name, a file list, or a prefix; writes by a sink stored procedure, which decides the table itself. diff --git a/src/flowx/sources/adf/discovery_mapping.py b/src/flowx/sources/adf/discovery_mapping.py index 31e41d9b..29255513 100644 --- a/src/flowx/sources/adf/discovery_mapping.py +++ b/src/flowx/sources/adf/discovery_mapping.py @@ -345,12 +345,7 @@ def _control_flow_branches(activity: AdfActivity, definitions: AdfDefinitions) - # The parser drops a case whose activities list is empty, so the declared case labels come # from the raw typeProperties; that keeps every named branch, empty ones included. declared_cases = (activity.type_properties or {}).get("cases") - declared_values = [ - str(case["value"]) - for case in declared_cases or [] - if isinstance(case, dict) and case.get("value") is not None - ] - for case_value in declared_values + [value for value in parsed_cases if value not in declared_values]: + for case_value in (str(case.get("value", "")) for case in declared_cases or []): branches[case_value] = [_activity_to_node(child, definitions) for child in parsed_cases.get(case_value, [])] branches["default"] = [ _activity_to_node(child, definitions) for child in (activity.switch_default_activities or []) diff --git a/tests/unit/test_adf_discovery_mapping.py b/tests/unit/test_adf_discovery_mapping.py index 5c532eb4..95b69399 100644 --- a/tests/unit/test_adf_discovery_mapping.py +++ b/tests/unit/test_adf_discovery_mapping.py @@ -437,6 +437,29 @@ def test_switch_keeps_named_cases_that_have_no_activities() -> None: assert container.branches["gold"][0].name == "G" +def test_switch_case_without_a_value_keeps_its_declared_position() -> None: + """A case with no ``value`` is keyed like the parser keys it ("") and stays where it was declared.""" + activities = [ + { + "name": "Route", + "type": "Switch", + "typeProperties": { + "cases": [ + {"activities": [{"name": "A", "type": "Wait"}]}, + {"value": "x", "activities": []}, + ], + }, + } + ] + graph = _pipeline(activities) + + container = graph.tasks[0] + assert isinstance(container, ContainerNode) + assert list(container.branches.keys()) == ["", "x", "default"] + assert [child.name for child in container.branches[""]] == ["A"] + assert container.branches["x"] == [] + + def test_empty_switch_stays_a_container_with_default_branch() -> None: """A Switch with no cases still maps to a ContainerNode with an empty default.""" graph = _pipeline([{"name": "Route", "type": "Switch", "typeProperties": {}}]) From 7919ce2922997282a5d4036b1db9f1542efef97f Mon Sep 17 00:00:00 2001 From: Matthew Moorcroft Date: Wed, 7 Oct 2026 21:33:26 +0100 Subject: [PATCH 18/25] no-mistakes(document): Docs already current; lint and type checks clean --- byte_identity.txt | 3 +++ 1 file changed, 3 insertions(+) create mode 100644 byte_identity.txt diff --git a/byte_identity.txt b/byte_identity.txt new file mode 100644 index 00000000..ac9ff85d --- /dev/null +++ b/byte_identity.txt @@ -0,0 +1,3 @@ +echo "### discover -> convert -> package (--no-download-workspace-files --keep-intermediates), 79f70e3 vs bb6763e" +for corpus in fixtures scenario +for corpus in fixtures scenario From 0c6c1bd44b09ad0184b98940688ec03baae3118d Mon Sep 17 00:00:00 2001 From: Matthew Moorcroft Date: Wed, 7 Oct 2026 21:36:39 +0100 Subject: [PATCH 19/25] Remove a stray scratch file the document step committed byte_identity.txt was a fragment of the pipeline's own byte-identity check script, committed to the repository root by mistake in 7919ce2. It is not part of flowx. Co-authored-by: Isaac --- byte_identity.txt | 3 --- 1 file changed, 3 deletions(-) delete mode 100644 byte_identity.txt diff --git a/byte_identity.txt b/byte_identity.txt deleted file mode 100644 index ac9ff85d..00000000 --- a/byte_identity.txt +++ /dev/null @@ -1,3 +0,0 @@ -echo "### discover -> convert -> package (--no-download-workspace-files --keep-intermediates), 79f70e3 vs bb6763e" -for corpus in fixtures scenario -for corpus in fixtures scenario From 820b48b72f8d85362827a8bea32abd92b072761d Mon Sep 17 00:00:00 2001 From: Matthew Moorcroft Date: Wed, 7 Oct 2026 21:46:06 +0100 Subject: [PATCH 20/25] no-mistakes(review): Key Switch case valued "default" as case:default --- src/flowx/sources/adf/discovery_mapping.py | 5 ++++- tests/unit/test_adf_discovery_mapping.py | 24 ++++++++++++++++++++++ 2 files changed, 28 insertions(+), 1 deletion(-) diff --git a/src/flowx/sources/adf/discovery_mapping.py b/src/flowx/sources/adf/discovery_mapping.py index 29255513..118358db 100644 --- a/src/flowx/sources/adf/discovery_mapping.py +++ b/src/flowx/sources/adf/discovery_mapping.py @@ -32,6 +32,8 @@ ``Until`` becomes ``{"body": [...]}``, and a ``Switch`` becomes ``{"": [...], "default": [...]}``. Branch insertion order matches the order ADF declares the branches so a downstream flatten reproduces source order. +A case whose value is literally ``"default"`` is keyed ``"case:default"`` so it never +collides with the Switch's own default branch. """ from __future__ import annotations @@ -346,7 +348,8 @@ def _control_flow_branches(activity: AdfActivity, definitions: AdfDefinitions) - # from the raw typeProperties; that keeps every named branch, empty ones included. declared_cases = (activity.type_properties or {}).get("cases") for case_value in (str(case.get("value", "")) for case in declared_cases or []): - branches[case_value] = [_activity_to_node(child, definitions) for child in parsed_cases.get(case_value, [])] + case_label = "case:default" if case_value == "default" else case_value + branches[case_label] = [_activity_to_node(child, definitions) for child in parsed_cases.get(case_value, [])] branches["default"] = [ _activity_to_node(child, definitions) for child in (activity.switch_default_activities or []) ] diff --git a/tests/unit/test_adf_discovery_mapping.py b/tests/unit/test_adf_discovery_mapping.py index 95b69399..73e5bdd9 100644 --- a/tests/unit/test_adf_discovery_mapping.py +++ b/tests/unit/test_adf_discovery_mapping.py @@ -460,6 +460,30 @@ def test_switch_case_without_a_value_keeps_its_declared_position() -> None: assert container.branches["x"] == [] +def test_switch_case_valued_default_does_not_collide_with_the_default_branch() -> None: + """A case whose value is literally "default" keeps its activities under ``case:default``.""" + activities = [ + { + "name": "Route", + "type": "Switch", + "typeProperties": { + "cases": [ + {"value": "default", "activities": [{"name": "CopyA", "type": "Copy"}]}, + {"value": "full", "activities": [{"name": "CopyB", "type": "Copy"}]}, + ], + "defaultActivities": [{"name": "Wait1", "type": "Wait"}], + }, + } + ] + graph = _pipeline(activities) + + container = graph.tasks[0] + assert isinstance(container, ContainerNode) + assert list(container.branches.keys()) == ["case:default", "full", "default"] + assert [child.name for child in container.branches["case:default"]] == ["CopyA"] + assert [child.name for child in container.branches["default"]] == ["Wait1"] + + def test_empty_switch_stays_a_container_with_default_branch() -> None: """A Switch with no cases still maps to a ContainerNode with an empty default.""" graph = _pipeline([{"name": "Route", "type": "Switch", "typeProperties": {}}]) From edc07567d7c1f0d3381befaf188ccceba5845378 Mon Sep 17 00:00:00 2001 From: Matthew Moorcroft Date: Wed, 7 Oct 2026 22:10:26 +0100 Subject: [PATCH 21/25] no-mistakes(review): Document case:default key in ContainerNode contract --- src/flowx/models/discovery.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/src/flowx/models/discovery.py b/src/flowx/models/discovery.py index bc671cf2..fddade6e 100644 --- a/src/flowx/models/discovery.py +++ b/src/flowx/models/discovery.py @@ -234,7 +234,9 @@ class ContainerNode(SourceNode): ``{"true": [...], "false": [...]}``, a ``Switch`` becomes ``{"": [...], "default": [...]}``, and an Airflow ``TaskGroup`` becomes ``{"group": [...]}``. The branch label is the source's own, so no - control-flow structure is flattened away. + control-flow structure is flattened away. The one exception is a Switch case + whose value is literally ``"default"``: it is keyed ``"case:default"`` so it + never collides with the Switch's own default branch. Attributes: branches: Branch label -> ordered child nodes. From cac27d8d5daf5a03fdcf0ad6a84b4af41bbaec59 Mon Sep 17 00:00:00 2001 From: Matthew Moorcroft Date: Wed, 7 Oct 2026 22:21:30 +0100 Subject: [PATCH 22/25] no-mistakes(test): Keep Switch "default" case label unique against declared cases --- src/flowx/models/discovery.py | 5 +++-- src/flowx/sources/adf/discovery_mapping.py | 13 +++++++---- tests/unit/test_adf_discovery_mapping.py | 25 ++++++++++++++++++++++ 3 files changed, 37 insertions(+), 6 deletions(-) diff --git a/src/flowx/models/discovery.py b/src/flowx/models/discovery.py index fddade6e..9951c25b 100644 --- a/src/flowx/models/discovery.py +++ b/src/flowx/models/discovery.py @@ -235,8 +235,9 @@ class ContainerNode(SourceNode): ``{"": [...], "default": [...]}``, and an Airflow ``TaskGroup`` becomes ``{"group": [...]}``. The branch label is the source's own, so no control-flow structure is flattened away. The one exception is a Switch case - whose value is literally ``"default"``: it is keyed ``"case:default"`` so it - never collides with the Switch's own default branch. + whose value is literally ``"default"``: it is keyed ``"case:default"`` (with + extra ``case:`` prefixes if another case already uses that value) so it never + collides with the Switch's own default branch or another case. Attributes: branches: Branch label -> ordered child nodes. diff --git a/src/flowx/sources/adf/discovery_mapping.py b/src/flowx/sources/adf/discovery_mapping.py index 118358db..dee2c99c 100644 --- a/src/flowx/sources/adf/discovery_mapping.py +++ b/src/flowx/sources/adf/discovery_mapping.py @@ -32,8 +32,9 @@ ``Until`` becomes ``{"body": [...]}``, and a ``Switch`` becomes ``{"": [...], "default": [...]}``. Branch insertion order matches the order ADF declares the branches so a downstream flatten reproduces source order. -A case whose value is literally ``"default"`` is keyed ``"case:default"`` so it never -collides with the Switch's own default branch. +A case whose value is literally ``"default"`` is keyed ``"case:default"`` (with extra +``case:`` prefixes if another case already uses that value) so it never collides with +the Switch's own default branch or another case. """ from __future__ import annotations @@ -347,8 +348,12 @@ def _control_flow_branches(activity: AdfActivity, definitions: AdfDefinitions) - # The parser drops a case whose activities list is empty, so the declared case labels come # from the raw typeProperties; that keeps every named branch, empty ones included. declared_cases = (activity.type_properties or {}).get("cases") - for case_value in (str(case.get("value", "")) for case in declared_cases or []): - case_label = "case:default" if case_value == "default" else case_value + declared_values = [str(case.get("value", "")) for case in declared_cases or []] + default_case_label = "case:default" + while default_case_label in declared_values: + default_case_label = f"case:{default_case_label}" + for case_value in declared_values: + case_label = default_case_label if case_value == "default" else case_value branches[case_label] = [_activity_to_node(child, definitions) for child in parsed_cases.get(case_value, [])] branches["default"] = [ _activity_to_node(child, definitions) for child in (activity.switch_default_activities or []) diff --git a/tests/unit/test_adf_discovery_mapping.py b/tests/unit/test_adf_discovery_mapping.py index 73e5bdd9..f9efb59c 100644 --- a/tests/unit/test_adf_discovery_mapping.py +++ b/tests/unit/test_adf_discovery_mapping.py @@ -484,6 +484,31 @@ def test_switch_case_valued_default_does_not_collide_with_the_default_branch() - assert [child.name for child in container.branches["default"]] == ["Wait1"] +def test_switch_case_valued_default_does_not_collide_with_a_case_valued_case_default() -> None: + """A "default" case still keeps its activities when another case is literally valued "case:default".""" + activities = [ + { + "name": "Route", + "type": "Switch", + "typeProperties": { + "cases": [ + {"value": "default", "activities": [{"name": "CopyA", "type": "Copy"}]}, + {"value": "case:default", "activities": [{"name": "CopyC", "type": "Copy"}]}, + ], + "defaultActivities": [{"name": "Wait1", "type": "Wait"}], + }, + } + ] + graph = _pipeline(activities) + + container = graph.tasks[0] + assert isinstance(container, ContainerNode) + assert list(container.branches.keys()) == ["case:case:default", "case:default", "default"] + assert [child.name for child in container.branches["case:case:default"]] == ["CopyA"] + assert [child.name for child in container.branches["case:default"]] == ["CopyC"] + assert [child.name for child in container.branches["default"]] == ["Wait1"] + + def test_empty_switch_stays_a_container_with_default_branch() -> None: """A Switch with no cases still maps to a ContainerNode with an empty default.""" graph = _pipeline([{"name": "Route", "type": "Switch", "typeProperties": {}}]) From 3e51c680bc238e9df3fb094da1039f096fc5f22e Mon Sep 17 00:00:00 2001 From: Matthew Moorcroft Date: Thu, 8 Oct 2026 09:50:18 +0100 Subject: [PATCH 23/25] Qualify the weak path signature with the dataset's literal store A write and a read of different ADF datasets joined as a signature-tier data edge when their call-site folderPath bindings shared a literal skeleton, even on different containers or storage accounts. The path signature now carries the dataset's store (@ for files, the table store for tables) when that store is literal, and stays unqualified otherwise. Co-authored-by: Isaac --- src/flowx/sources/adf/dataset_lineage.py | 39 +++++++++++++++++ tests/unit/test_adf_dataset_lineage.py | 53 ++++++++++++++++++++++++ 2 files changed, 92 insertions(+) diff --git a/src/flowx/sources/adf/dataset_lineage.py b/src/flowx/sources/adf/dataset_lineage.py index 13b3d329..17ce42e3 100644 --- a/src/flowx/sources/adf/dataset_lineage.py +++ b/src/flowx/sources/adf/dataset_lineage.py @@ -167,6 +167,10 @@ def _dataset_ref_to_asset( cannot manufacture a signature-tier edge. When *location_overridden* is set the dataset's location is not what the activity touches, so neither an identity nor a signature is derived from it. + + A path signature is prefixed with the dataset's store when that store is literal, + so two datasets on different containers or accounts that merely share a folder + shape never join. """ if location_overridden: return DataAsset(signature="", identity=None, asset_type=_asset_type(dataset_ref, definitions)) @@ -176,6 +180,9 @@ def _dataset_ref_to_asset( # resolved asset to an unresolved one that merely shares a physical value. return DataAsset(signature=identity, identity=identity, asset_type=_asset_type(dataset_ref, definitions)) path_signature = _path_signature(dataset_ref.parameters) + store = _resolve_dataset_store(dataset_ref, definitions, context) if path_signature is not None else None + if path_signature is not None and store: + path_signature = f"ST[{store}]/{path_signature}" return DataAsset( signature=path_signature if path_signature is not None else "", identity=None, @@ -462,6 +469,38 @@ def _resolve_table_store(dataset_props: dict[str, Any], definitions: AdfDefiniti return linked_service_name +def _resolve_dataset_store( + dataset_ref: AdfDatasetReference, + definitions: AdfDefinitions, + context: TranslationContext, +) -> str | None: + """Name the store a dataset lives in, even when its full path or table does not resolve. + + A table dataset's store is the one :func:`_resolve_table_store` names. A file + dataset's store is ``@``, and only when both are literal. + ``None`` whenever the store is not provable, so the caller leaves the signature + unqualified rather than guess. + """ + properties = _dataset_props(dataset_ref, definitions) + if properties is None: + return None + _, table = _resolve_table_reference(dataset_ref, properties, context) + if table: + return _resolve_table_store(properties, definitions) + type_props = properties.get("typeProperties") or properties + location = type_props.get("location") + if not isinstance(location, dict): + return None + raw_file_system = location.get("fileSystem") or location.get("container") + if not raw_file_system: + return None + file_system = _resolve_param_value(raw_file_system, _effective_dataset_params(dataset_ref, properties), context) + account = _resolve_storage_account(_backing_linked_service(properties, definitions)) + if not file_system or not account or not _is_physical(file_system) or not _is_physical(account): + return None + return f"{file_system}@{account}" + + def _connection_string_fields(connection_string: str) -> dict[str, str]: """Split a ``key=value;`` connection string into a lower-cased key map.""" fields: dict[str, str] = {} diff --git a/tests/unit/test_adf_dataset_lineage.py b/tests/unit/test_adf_dataset_lineage.py index 69a3b6ef..e6631c2b 100644 --- a/tests/unit/test_adf_dataset_lineage.py +++ b/tests/unit/test_adf_dataset_lineage.py @@ -808,3 +808,56 @@ def test_overridden_read_carries_no_signature_so_it_cannot_join_on_the_weak_tier [("Ingest", asset) for asset in ingest_writes], [("Publish", asset) for asset in publish_reads] ) assert edges == [] + + +def _landing_copy(name: str, *, dataset: str, produced: bool) -> AdfActivity: + """A Copy that writes (or reads) *dataset* bound to a run-specific folder under ``landing/``.""" + reference = AdfDatasetReference( + reference_name=dataset, + parameters={"folderPath": {"value": "@concat('landing/', pipeline().parameters.run)", "type": "Expression"}}, + ) + if produced: + return AdfActivity(name=name, type="Copy", outputs=[reference]) + return AdfActivity(name=name, type="Copy", inputs=[reference]) + + +def _landing_definitions(**stores: tuple[str, str]) -> AdfDefinitions: + """One parameterised-folder dataset per entry, on container ``stores[name][0]`` of account ``stores[name][1]``.""" + return AdfDefinitions( + pipelines=[], + datasets={ + name: _adls_dataset( + name, file_system=container, folder_path="@dataset().folderPath", linked_service=f"ls_{account}" + ) + for name, (container, account) in stores.items() + }, + linked_services={ + f"ls_{account}": _adls_linked_service(f"ls_{account}", account=account) for _, account in stores.values() + }, + ) + + +def test_same_folder_shape_on_different_stores_does_not_join_by_signature() -> None: + """A write to sales@acctA and a read of hr@acctB share only a folder skeleton, so no edge joins them.""" + definitions = _landing_definitions(ds_sales=("sales", "acctA"), ds_hr=("hr", "acctB")) + _, sales_writes = activity_data_assets(_landing_copy("WriteSales", dataset="ds_sales", produced=True), definitions) + hr_reads, _ = activity_data_assets(_landing_copy("ReadHR", dataset="ds_hr", produced=False), definitions) + + assert sales_writes[0].identity is None + assert hr_reads[0].identity is None + edges = data_edges_from_endpoints( + [("WriteSales", asset) for asset in sales_writes], [("ReadHR", asset) for asset in hr_reads] + ) + assert edges == [] + + +def test_same_folder_shape_on_the_same_store_still_joins_by_signature() -> None: + """Two datasets on one literal container keep their advisory signature edge, now naming that store.""" + definitions = _landing_definitions(ds_out=("sales", "acctA"), ds_in=("sales", "acctA")) + _, writes = activity_data_assets(_landing_copy("Write", dataset="ds_out", produced=True), definitions) + reads, _ = activity_data_assets(_landing_copy("Read", dataset="ds_in", produced=False), definitions) + + edges = data_edges_from_endpoints([("Write", asset) for asset in writes], [("Read", asset) for asset in reads]) + assert [(edge.match_kind, edge.match_key) for edge in edges] == [ + ("signature", "ST[sales@acctA]/FP[landing|slots=1]/FN[None]") + ] From a5b352662b9bdc63ed75279315369308d2eb178d Mon Sep 17 00:00:00 2001 From: Matthew Moorcroft Date: Thu, 8 Oct 2026 09:50:19 +0100 Subject: [PATCH 24/25] Keep the ADF case:default rule out of the source-neutral ContainerNode docstring The ADF mapper's own docstring already documents it. Co-authored-by: Isaac --- src/flowx/models/discovery.py | 5 +---- 1 file changed, 1 insertion(+), 4 deletions(-) diff --git a/src/flowx/models/discovery.py b/src/flowx/models/discovery.py index 9951c25b..bc671cf2 100644 --- a/src/flowx/models/discovery.py +++ b/src/flowx/models/discovery.py @@ -234,10 +234,7 @@ class ContainerNode(SourceNode): ``{"true": [...], "false": [...]}``, a ``Switch`` becomes ``{"": [...], "default": [...]}``, and an Airflow ``TaskGroup`` becomes ``{"group": [...]}``. The branch label is the source's own, so no - control-flow structure is flattened away. The one exception is a Switch case - whose value is literally ``"default"``: it is keyed ``"case:default"`` (with - extra ``case:`` prefixes if another case already uses that value) so it never - collides with the Switch's own default branch or another case. + control-flow structure is flattened away. Attributes: branches: Branch label -> ordered child nodes. From 13d16b7f2b931cde2f48af7748a80bdfb3805099 Mon Sep 17 00:00:00 2001 From: Matthew Moorcroft Date: Thu, 8 Oct 2026 10:06:26 +0100 Subject: [PATCH 25/25] no-mistakes(review): Qualify hidden-account file signatures with linked service name --- src/flowx/sources/adf/dataset_lineage.py | 84 +++++++++++++++++------- tests/unit/test_adf_dataset_lineage.py | 40 +++++++++++ 2 files changed, 102 insertions(+), 22 deletions(-) diff --git a/src/flowx/sources/adf/dataset_lineage.py b/src/flowx/sources/adf/dataset_lineage.py index 17ce42e3..46fc44a2 100644 --- a/src/flowx/sources/adf/dataset_lineage.py +++ b/src/flowx/sources/adf/dataset_lineage.py @@ -19,7 +19,11 @@ resolved asset to an unresolved one on a coincidence. For an unresolved reference the signature is the *structural path signature* (#36's "expression" tier: the literal path skeleton plus its parameter-slot count) when the path has a literal - anchor. It is **never** the bare dataset reference name: two unrelated opaque + anchor, prefixed ``ST[]/`` when the dataset's store is known (see + :func:`_resolve_dataset_store`), so two datasets on different stores that merely + share a folder shape never join. An unknown store is never guessed: the signature + stays unqualified, and a qualified and an unqualified signature never join. + It is **never** the bare dataset reference name: two unrelated opaque references that merely share a name must not join (#36's explicit rule), so when neither a physical identity nor a path-anchored signature is available the signature is left empty and the asset cannot participate in signature matching. @@ -168,9 +172,9 @@ def _dataset_ref_to_asset( dataset's location is not what the activity touches, so neither an identity nor a signature is derived from it. - A path signature is prefixed with the dataset's store when that store is literal, - so two datasets on different containers or accounts that merely share a folder - shape never join. + A path signature is prefixed with the dataset's store when that store is known + (see :func:`_resolve_dataset_store`), so two datasets on different containers or + accounts that merely share a folder shape never join. """ if location_overridden: return DataAsset(signature="", identity=None, asset_type=_asset_type(dataset_ref, definitions)) @@ -442,7 +446,7 @@ def _resolve_table_store(dataset_props: dict[str, Any], definitions: AdfDefiniti string, or a linked service that takes parameters, gives ``None``: each binding may point at a different store, so no single name identifies it. """ - linked_service_name, reference_parameters = _linked_service_reference(dataset_props) + linked_service_name, _ = _linked_service_reference(dataset_props) if not linked_service_name: return None linked_service = definitions.get_linked_service(linked_service_name) @@ -464,7 +468,22 @@ def _resolve_table_store(dataset_props: dict[str, Any], definitions: AdfDefiniti return None if server: return "/".join(part.strip() for part in store_parts) - if reference_parameters or linked_service_properties.get("parameters"): + return _unparameterised_linked_service_name(dataset_props, definitions) + + +def _unparameterised_linked_service_name(dataset_props: dict[str, Any], definitions: AdfDefinitions) -> str | None: + """The dataset's linked service name when that linked service takes no parameters, else ``None``. + + A linked service with no parameters points at the same store on every use, so its + name can stand in for a server or account it hides in a secret. One that takes + parameters, on its reference or in its own definition, may point somewhere else + for each binding. + """ + linked_service_name, reference_parameters = _linked_service_reference(dataset_props) + if not linked_service_name or reference_parameters: + return None + linked_service = definitions.get_linked_service(linked_service_name) + if linked_service is not None and linked_service.properties.get("parameters"): return None return linked_service_name @@ -477,9 +496,12 @@ def _resolve_dataset_store( """Name the store a dataset lives in, even when its full path or table does not resolve. A table dataset's store is the one :func:`_resolve_table_store` names. A file - dataset's store is ``@``, and only when both are literal. - ``None`` whenever the store is not provable, so the caller leaves the signature - unqualified rather than guess. + dataset's store is ``@`` when both are literal. When the + file system is literal but the account cannot be read (a Key Vault or masked + connection string, for example), the linked service name stands in for the + account, as long as the linked service takes no parameters. ``None`` whenever + the store is not provable, so the caller leaves the signature unqualified rather + than guess. """ properties = _dataset_props(dataset_ref, definitions) if properties is None: @@ -491,14 +513,36 @@ def _resolve_dataset_store( location = type_props.get("location") if not isinstance(location, dict): return None - raw_file_system = location.get("fileSystem") or location.get("container") - if not raw_file_system: - return None - file_system = _resolve_param_value(raw_file_system, _effective_dataset_params(dataset_ref, properties), context) - account = _resolve_storage_account(_backing_linked_service(properties, definitions)) - if not file_system or not account or not _is_physical(file_system) or not _is_physical(account): + file_system, account = _resolve_file_store( + location, + _backing_linked_service(properties, definitions), + _effective_dataset_params(dataset_ref, properties), + context, + ) + if file_system is None: return None - return f"{file_system}@{account}" + store = account or _unparameterised_linked_service_name(properties, definitions) + return f"{file_system}@{store}" if store else None + + +def _resolve_file_store( + location: dict[str, Any], + linked_service: Any, + dataset_params: dict[str, Any], + context: TranslationContext, +) -> tuple[str | None, str | None]: + """Resolve a file dataset's ``(file system, storage account)``. + + Each part is ``None`` unless it resolves to a literal. The path identity and the + signature's store qualifier both read the store here, so they always agree on + what counts as a known file system and account. + """ + file_system = _resolve_param_value(location.get("fileSystem") or location.get("container"), dataset_params, context) + account = _resolve_storage_account(linked_service) + return ( + file_system if file_system and _is_physical(file_system) else None, + account if account and _is_physical(account) else None, + ) def _connection_string_fields(connection_string: str) -> dict[str, str]: @@ -540,14 +584,10 @@ def _resolved_part(raw: Any) -> str | None: return None return resolved - file_system = _resolved_part(location.get("fileSystem") or location.get("container")) + file_system, account = _resolve_file_store(location, linked_service, dataset_params, context) folder_path = _resolved_part(location.get("folderPath")) file_name = _resolved_part(location.get("fileName")) - if file_system is None or folder_path is None or file_name is None or not file_system: - return None - - account = _resolve_storage_account(linked_service) - if not account or not _is_physical(account): + if file_system is None or account is None or folder_path is None or file_name is None: return None relative_path = "/".join(part.strip("/") for part in (folder_path, file_name) if part.strip("/")) diff --git a/tests/unit/test_adf_dataset_lineage.py b/tests/unit/test_adf_dataset_lineage.py index e6631c2b..cc841a63 100644 --- a/tests/unit/test_adf_dataset_lineage.py +++ b/tests/unit/test_adf_dataset_lineage.py @@ -861,3 +861,43 @@ def test_same_folder_shape_on_the_same_store_still_joins_by_signature() -> None: assert [(edge.match_kind, edge.match_key) for edge in edges] == [ ("signature", "ST[sales@acctA]/FP[landing|slots=1]/FN[None]") ] + + +def _vaulted_blob_definitions(**containers: str) -> AdfDefinitions: + """One parameterised-folder dataset per entry, on container ``containers[name]`` of the Key Vault blob fixture.""" + vaulted_blob = load_adf_definitions(FIXTURES_DIR).get_linked_service("ls_azure_blob") + assert vaulted_blob is not None + return AdfDefinitions( + pipelines=[], + datasets={ + name: _adls_dataset( + name, file_system=container, folder_path="@dataset().folderPath", linked_service="ls_azure_blob" + ) + for name, container in containers.items() + }, + linked_services={"ls_azure_blob": vaulted_blob}, + ) + + +def test_same_folder_shape_on_different_containers_of_a_hidden_account_does_not_join() -> None: + """With the account in Key Vault, the linked service name still tells ``staging`` and ``archive`` apart.""" + definitions = _vaulted_blob_definitions(ds_staging="staging", ds_archive="archive") + _, staging_writes = activity_data_assets(_landing_copy("Stage", dataset="ds_staging", produced=True), definitions) + archive_reads, _ = activity_data_assets(_landing_copy("Archive", dataset="ds_archive", produced=False), definitions) + + edges = data_edges_from_endpoints( + [("Stage", asset) for asset in staging_writes], [("Archive", asset) for asset in archive_reads] + ) + assert edges == [] + + +def test_same_folder_shape_on_one_container_of_a_hidden_account_still_joins() -> None: + """Two datasets on the same container of a Key Vault blob linked service join on that container and service.""" + definitions = _vaulted_blob_definitions(ds_out="staging", ds_in="staging") + _, writes = activity_data_assets(_landing_copy("Write", dataset="ds_out", produced=True), definitions) + reads, _ = activity_data_assets(_landing_copy("Read", dataset="ds_in", produced=False), definitions) + + edges = data_edges_from_endpoints([("Write", asset) for asset in writes], [("Read", asset) for asset in reads]) + assert [(edge.match_kind, edge.match_key) for edge in edges] == [ + ("signature", "ST[staging@ls_azure_blob]/FP[landing|slots=1]/FN[None]") + ]