diff --git a/AGENTS.md b/AGENTS.md index e1ae8c4..8befda1 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -56,7 +56,7 @@ All three phases write into one shared `` (default `./flowx_output`) the DAB bundle at the top level, kept artifacts under `metadata/`, and transient intermediates under `.work/` (pruned by `package`). -1. **Discover** -- Parse ADF JSON from UC volumes -> typed AST -> `metadata/inventory.json` + `metadata/profile_report.csv` + verbatim `metadata/.arm.json` +1. **Discover** -- Parse ADF JSON from UC volumes -> typed AST -> `metadata/source_graphs.json` (hashed) -> `metadata/inventory.json` + `metadata/profile_report.csv` + verbatim `metadata/.arm.json` 2. **Convert** -- Registry dispatch + topological sort -> Pipeline IR (deterministic + agentic gaps); transient report at `.work/translation_report.json` 3. **Package** -- IR -> DAB YAML + generated notebooks + setup scripts; prunes `.work/` diff --git a/README.md b/README.md index ea3e7e8..42d016a 100644 --- a/README.md +++ b/README.md @@ -257,6 +257,7 @@ flowx_output/ SETUP.md # Setup instructions (package) metadata/ inventory.json # discover: activity inventory + source_graphs.json # discover (ADF): saved source graphs the inventory is built from profile_report.csv # discover: per-pipeline complexity report .arm.json # discover: verbatim original ADF/ARM source configuration.json # modify: collected configuration answers diff --git a/skills/flowx-discover/SKILL.md b/skills/flowx-discover/SKILL.md index b27e6f0..6c3e9b5 100644 --- a/skills/flowx-discover/SKILL.md +++ b/skills/flowx-discover/SKILL.md @@ -65,6 +65,7 @@ All under the shared `/metadata/` folder: | File | Description | |---|---| | `metadata/inventory.json` | Classified activity inventory for the convert phase | +| `metadata/source_graphs.json` | (ADF) Saved, content-hashed source graphs (activities, lineage, detected motifs); the inventory is built from it and records its hash as `source_graphs_sha256` | | `metadata/profile_report.csv` | Per-pipeline complexity report (counts + T-shirt size) | | `metadata/.arm.json` | (ADF) Verbatim original source for each pipeline (provenance) | diff --git a/skills/flowx-discover/sources/adf.md b/skills/flowx-discover/sources/adf.md index 9997e27..f79205c 100644 --- a/skills/flowx-discover/sources/adf.md +++ b/skills/flowx-discover/sources/adf.md @@ -51,16 +51,24 @@ Read `/metadata/inventory.json`: { "name": "PipelineName", "activities": [ - {"name": "CopyFromBlob", "type": "Copy", "strategy": "deterministic", "translator": "copy.py"}, - {"name": "RunDataFlow", "type": "ExecuteDataFlow", "strategy": "agentic"} + {"name": "CopyFromBlob", "type": "Copy", "strategy": "deterministic", "task_key": "CopyFromBlob"}, + {"name": "RunDataFlow", "type": "ExecuteDataFlow", "strategy": "agentic", "task_key": "RunDataFlow"} ] } ], "summary": {"pipeline_count": 12, "activity_count": 47, "deterministic_count": 35, - "agentic_count": 10, "unsupported_count": 2, "coverage_pct": 95.7} + "agentic_count": 10, "unsupported_count": 2, "coverage_pct": 95.7}, + "source_graphs_sha256": "" } ``` +Each activity's `task_key` equals its ADF activity name. `source_graphs_sha256` names the saved +`metadata/source_graphs.json` the inventory was built from. + +The inventory deliberately has no top-level `generated_at` timestamp any more, so the same export +always produces byte-identical `inventory.json` and any hash taken over it stays stable. Use the +file's modification time if you need to know when discover ran. + ## Step 4b — Review the complexity report `/metadata/profile_report.csv` has one row per pipeline: `pipeline`, `activities`, diff --git a/src/flowx/bundler/dab_writer.py b/src/flowx/bundler/dab_writer.py index 436900d..c72fe5a 100644 --- a/src/flowx/bundler/dab_writer.py +++ b/src/flowx/bundler/dab_writer.py @@ -26,6 +26,7 @@ from flowx.bundler.notebook_writer import write_notebooks from flowx.bundler.prereqs_writer import ManualParameter, build_prereqs, render_setup_md from flowx.bundler.setup_generator import generate_setup_tasks +from flowx.ir_serde import data_asset_from_dict, lineage_from_dict from flowx.models.dab import DabNotebook from flowx.models.ir import ( Activity, @@ -2021,6 +2022,7 @@ def pipeline_dict_to_ir(pipeline_dict: dict[str, Any]) -> tuple[Pipeline, list[d ): raise ValueError(f"Invalid Pipeline email notification entry: {event!r}") email_notifications[str(event)] = list(recipients) + raw_lineage = pipeline_dict.get("lineage") pipeline = Pipeline( name=pipeline_dict.get("name", "unknown"), @@ -2037,6 +2039,7 @@ def pipeline_dict_to_ir(pipeline_dict: dict[str, Any]) -> tuple[Pipeline, list[d reconciliation_status=pipeline_dict.get("reconciliation_status"), migration_status=pipeline_dict.get("migration_status", "included"), audit=dict(pipeline_dict.get("audit") or {}), + lineage=lineage_from_dict(raw_lineage) if raw_lineage else None, ) return pipeline, parameters @@ -2258,9 +2261,10 @@ def _reconstruct_ir(task_ir: dict[str, Any]) -> Activity: bridge_required_parameters=dict(task_ir.get("bridge_required_parameters") or {}), ) if task_type == "MotifActivity": + # motif_id arrives via ``base`` (Activity now owns the field); passing it again here + # would raise "multiple values for keyword argument 'motif_id'". return MotifActivity( **base, - motif_id=task_ir.get("motif_id", "unknown"), display_name=task_ir.get("display_name", base["name"]), databricks_replacement=task_ir.get("databricks_replacement", "notebook"), matched_activity_names=list(task_ir.get("matched_activity_names", [])), @@ -2311,6 +2315,9 @@ def _common_activity_kwargs(task_ir: dict[str, Any]) -> dict[str, Any]: "required_parameters": dict(task_ir.get("required_parameters") or {}), "compute_mode": task_ir.get("compute_mode"), "notifications": task_ir.get("notifications"), + "motif_id": task_ir.get("motif_id"), + "data_reads": [data_asset_from_dict(asset) for asset in task_ir.get("data_reads") or []], + "data_writes": [data_asset_from_dict(asset) for asset in task_ir.get("data_writes") or []], } diff --git a/src/flowx/discovery_serde.py b/src/flowx/discovery_serde.py index 7af5ce8..95e1779 100644 --- a/src/flowx/discovery_serde.py +++ b/src/flowx/discovery_serde.py @@ -16,9 +16,8 @@ resolvable physical ``identity`` when there is one (else ``None``), an always-present ``signature``, and an open ``asset_type`` that also covers non-physical / logical / value hand-offs (e.g. an Airflow XCom). A graph's derived -:class:`~flowx.models.ir.Lineage` block is serialised through ``ir_serde``'s -``lineage_to_dict`` for the same reason; its inverse (:func:`_lineage_from_dict`) -lives here because ``ir_serde`` ships only the forward direction. +:class:`~flowx.models.ir.Lineage` block goes through ``ir_serde``'s +``lineage_to_dict`` / ``lineage_from_dict`` pair for the same reason. Every node dict carries a ``node_type`` discriminator (the dataclass name) so a :class:`~flowx.models.discovery.ContainerNode` or @@ -33,7 +32,7 @@ from pathlib import Path from typing import Any -from flowx.ir_serde import data_asset_from_dict, data_asset_to_dict, lineage_to_dict +from flowx.ir_serde import data_asset_from_dict, data_asset_to_dict, lineage_from_dict, lineage_to_dict from flowx.models.discovery import ( ContainerNode, GapNode, @@ -44,7 +43,6 @@ SourceGraph, SourceNode, ) -from flowx.models.ir import ControlEdge, DataEdge, Lineage, MotifAnnotation SOURCE_GRAPHS_FILENAME = "source_graphs.json" SOURCE_GRAPHS_CONTRACT_VERSION = "1" @@ -97,7 +95,7 @@ def source_graph_from_dict(raw: dict[str, Any]) -> SourceGraph: run_timeout_seconds=raw.get("run_timeout_seconds"), tags=list(raw.get("tags") or []), tasks=[_node_from_dict(node) for node in raw.get("tasks") or []], - lineage=_lineage_from_dict(lineage) if lineage else None, + lineage=lineage_from_dict(lineage) if lineage else None, properties=dict(raw.get("properties") or {}), extensions=dict(raw.get("extensions") or {}), raw=raw.get("raw"), @@ -188,48 +186,6 @@ def _canonical_sha256(value: Any) -> str: return hashlib.sha256(canonical.encode("utf-8")).hexdigest() -def _lineage_from_dict(raw: dict[str, Any]) -> Lineage: - """Rehydrate a :class:`Lineage` block from the dict ``ir_serde.lineage_to_dict`` emits. - - The inverse of that forward serialiser (which ``ir_serde`` does not itself - ship), so a discovery graph's lineage round-trips through this module. - """ - return Lineage( - control_edges=[ - ControlEdge( - source_workflow=edge.get("source_workflow", ""), - target_workflow=edge.get("target_workflow", ""), - via_task_key=edge.get("via_task_key", ""), - wait_for_completion=edge.get("wait_for_completion"), - resolved=bool(edge.get("resolved", True)), - ) - for edge in raw.get("control_edges") or [] - ], - data_edges=[ - DataEdge( - source_task_key=edge.get("source_task_key", ""), - target_task_key=edge.get("target_task_key", ""), - match_kind=edge.get("match_kind", ""), - match_key=edge.get("match_key", ""), - identity=edge.get("identity"), - asset_type=edge.get("asset_type"), - ) - for edge in raw.get("data_edges") or [] - ], - motifs=[ - MotifAnnotation( - motif_id=motif.get("motif_id", ""), - member_task_keys=list(motif.get("member_task_keys") or []), - display_name=motif.get("display_name"), - databricks_replacement=motif.get("databricks_replacement"), - notes=list(motif.get("notes") or []), - source_type_hint=motif.get("source_type_hint"), - ) - for motif in raw.get("motifs") or [] - ], - ) - - def _parameter_to_dict(spec: ParameterSpec) -> dict[str, Any]: result: dict[str, Any] = {} if spec.type is not None: diff --git a/src/flowx/ir_serde.py b/src/flowx/ir_serde.py index 8117322..3dbbe31 100644 --- a/src/flowx/ir_serde.py +++ b/src/flowx/ir_serde.py @@ -22,8 +22,10 @@ from flowx.models.ir import ( Activity, AppendVariableActivity, + ControlEdge, CopyActivity, DataAsset, + DataEdge, DbtFactoryActivity, DeleteActivity, ExecutePipelineActivity, @@ -83,6 +85,8 @@ def pipeline_to_dict(pipeline: Pipeline) -> dict[str, Any]: } if pipeline.translation_configuration is not None: result["translation_configuration"] = configuration_to_dict(pipeline.translation_configuration) + if pipeline.lineage is not None: + result["lineage"] = lineage_to_dict(pipeline.lineage) return result @@ -148,6 +152,48 @@ def lineage_to_dict(lineage: Lineage) -> dict[str, Any]: } +def lineage_from_dict(raw: dict[str, Any]) -> Lineage: + """Rehydrate a :class:`Lineage` block from the dict :func:`lineage_to_dict` emits. + + The one inverse shared by the translation report and the discovery graph, so a + lineage block reads back the same way wherever it was written. + """ + return Lineage( + control_edges=[ + ControlEdge( + source_workflow=edge.get("source_workflow", ""), + target_workflow=edge.get("target_workflow", ""), + via_task_key=edge.get("via_task_key", ""), + wait_for_completion=edge.get("wait_for_completion"), + resolved=bool(edge.get("resolved", True)), + ) + for edge in raw.get("control_edges") or [] + ], + data_edges=[ + DataEdge( + source_task_key=edge.get("source_task_key", ""), + target_task_key=edge.get("target_task_key", ""), + match_kind=edge.get("match_kind", ""), + match_key=edge.get("match_key", ""), + identity=edge.get("identity"), + asset_type=edge.get("asset_type"), + ) + for edge in raw.get("data_edges") or [] + ], + motifs=[ + MotifAnnotation( + motif_id=motif.get("motif_id", ""), + member_task_keys=list(motif.get("member_task_keys") or []), + display_name=motif.get("display_name"), + databricks_replacement=motif.get("databricks_replacement"), + notes=list(motif.get("notes") or []), + source_type_hint=motif.get("source_type_hint"), + ) + for motif in raw.get("motifs") or [] + ], + ) + + def _motif_annotation_to_dict(motif: MotifAnnotation) -> dict[str, Any]: """Serialise one motif annotation; ``source_type_hint`` is written only when set. @@ -226,6 +272,12 @@ def activity_to_dict(task: Activity) -> dict[str, Any]: task_dict["libraries"] = task.libraries if task.parameter_approximations: task_dict["parameter_approximations"] = task.parameter_approximations + if task.motif_id: + task_dict["motif_id"] = task.motif_id + if task.data_reads: + task_dict["data_reads"] = [data_asset_to_dict(asset) for asset in task.data_reads] + if task.data_writes: + task_dict["data_writes"] = [data_asset_to_dict(asset) for asset in task.data_writes] extra = activity_extra_fields(task) task_dict.update(extra) @@ -420,7 +472,7 @@ def activity_extra_fields(activity: Activity) -> dict[str, Any]: if activity.job_parameters: extra["job_parameters"] = activity.job_parameters case MotifActivity(): - extra["motif_id"] = activity.motif_id + # motif_id now lives on the Activity base and is serialised by activity_to_dict. extra["display_name"] = activity.display_name extra["databricks_replacement"] = activity.databricks_replacement extra["matched_activity_names"] = activity.matched_activity_names diff --git a/src/flowx/lineage.py b/src/flowx/lineage.py index b603481..f89f4d5 100644 --- a/src/flowx/lineage.py +++ b/src/flowx/lineage.py @@ -1,24 +1,76 @@ -"""Source-neutral lineage-edge cores. - -Primitive-level building blocks for deriving lineage edges. They take only -``task_key`` strings and :class:`~flowx.models.ir.DataAsset` values -- never a -``Pipeline`` or any ``Activity`` -- so the tier-matching, self-edge drop, and -dedup rules live in exactly one place. The discovery-graph derivation in -:mod:`flowx.discovery_lineage` gathers those primitives from a -:class:`~flowx.models.discovery.SourceGraph` and delegates here; the IR-facing -derivation that gathers them from a :class:`~flowx.models.ir.Pipeline` is added -separately in the convert->package lineage work so it stacks on this standard. - -They import nothing from ``sources/adf`` or ``sources/airflow``. Everything here -is pure: the functions read their inputs and return new edge lists; nothing is -mutated. +"""Source-neutral lineage derivation over the flowx Pipeline IR. + +These functions turn an already-translated :class:`~flowx.models.ir.Pipeline` +into its :class:`~flowx.models.ir.Lineage` block. They operate on IR primitives +only -- ``Activity`` subclasses, ``DataAsset``, ``task_key`` -- and import nothing +from ``sources/adf`` or ``sources/airflow`` so both front-ends share one code +path once they populate ``data_reads`` / ``data_writes`` / ``motif_id``. + +The tier-matching, self-edge drop, and dedup rules are factored into two +primitive-level cores -- :func:`control_edges_from_calls` and +:func:`data_edges_from_endpoints` -- that take only ``task_key`` strings and +:class:`DataAsset` values, never a ``Pipeline``. The IR entry points +(:func:`build_control_edges` / :func:`build_data_edges`) gather those primitives +from a pipeline and delegate, and the source-neutral discovery-graph derivation in +:mod:`flowx.discovery_lineage` gathers the same primitives from a +:class:`~flowx.models.discovery.SourceGraph` and delegates too, so both phases +join edges through exactly one implementation. + +Everything here is pure: the functions read the pipeline and return new edge +lists / a new :class:`Lineage`; nothing is mutated. :func:`with_lineage` attaches +a block by returning a *new* ``Pipeline`` rather than mutating the input, unlike +the in-place dependency rewrite in ``motifs/collapser.py``. """ from __future__ import annotations -from collections.abc import Iterable +import dataclasses +from collections.abc import Iterable, Iterator + +from flowx.models.ir import ( + Activity, + ControlEdge, + DataAsset, + DataEdge, + ExecutePipelineActivity, + ForEachActivity, + IfConditionActivity, + Lineage, + MotifActivity, + MotifAnnotation, + Pipeline, + RunJobActivity, + SwitchActivity, +) + + +def walk_activities(activities: list[Activity]) -> Iterator[Activity]: + """Yield every activity in *activities*, descending into control-flow containers. + + Recurses into ForEach inner activities, both If-condition branches, and every + Switch case plus its default branch, so a nested ExecutePipeline or a data + asset buried inside a Switch case is still reached. Motif ``original_activities`` + are intentionally not traversed: they are the pre-collapse originals kept for + reference, not live graph members. -from flowx.models.ir import ControlEdge, DataAsset, DataEdge + Args: + activities: Top-level (or already-nested) activity list to walk. + + Yields: + Each activity, container nodes included, in depth-first order. + """ + for activity in activities: + yield activity + match activity: + case ForEachActivity(): + yield from walk_activities(activity.inner_activities) + case IfConditionActivity(): + yield from walk_activities(activity.if_true_activities) + yield from walk_activities(activity.if_false_activities) + case SwitchActivity(): + for case_branch in activity.cases: + yield from walk_activities(case_branch.activities) + yield from walk_activities(activity.default_activities) def control_edges_from_calls( @@ -27,10 +79,10 @@ def control_edges_from_calls( ) -> list[ControlEdge]: """Assemble deduplicated control edges from raw invocation primitives. - The shared core behind the discovery-graph control-edge derivation (and the - IR-facing derivation added in the convert->package work): it owns the - self-edge drop, the dedup, and the unresolved-callee recording so those rules - live in exactly one place and every phase behaves identically. + The shared core behind :func:`build_control_edges` (IR) and the discovery-graph + control-edge derivation: it owns the self-edge drop, the dedup, and the + unresolved-callee recording so those rules live in exactly one place and both + phases behave identically. Args: source_workflow: Name of the calling workflow (pipeline / DAG). @@ -65,6 +117,38 @@ def control_edges_from_calls( return edges +def build_control_edges(pipeline: Pipeline) -> list[ControlEdge]: + """Derive cross-workflow invocation edges for a pipeline. + + Emits one :class:`ControlEdge` per invoking activity -- an + ``ExecutePipelineActivity`` (ADF) or a ``RunJobActivity`` (Airflow) -- found + anywhere in the pipeline, including inside ForEach / If / Switch containers + (fan-out is preserved: each call site is its own edge). Edges whose callee + equals the caller are dropped (no self-edges), and identical edges are + collapsed (no duplicates). An unresolved callee is recorded with + ``resolved=False`` rather than dropped. A ``RunJobActivity`` that runs an + existing job by ID emits no edge. + + Args: + pipeline: The translated pipeline IR. + + Returns: + Deduplicated list of control edges, in first-seen order. + """ + + def _calls() -> Iterator[tuple[str, bool | None, str]]: + for activity in walk_activities(pipeline.tasks): + match activity: + case ExecutePipelineActivity(): + yield activity.pipeline_name or "", activity.wait_on_completion, activity.task_key + case RunJobActivity() if not activity.existing_job_id: + # A run-now of an existing job by ID targets a job outside this source, and its + # job_name is the caller's own task key, so it is not a cross-workflow edge. + yield activity.job_name or "", None, activity.task_key + + return control_edges_from_calls(pipeline.name, _calls()) + + def _match_assets(producer: DataAsset, consumer: DataAsset) -> tuple[str, str, str | None] | None: """Decide whether a written asset hands off to a read asset, and how. @@ -96,10 +180,9 @@ def data_edges_from_endpoints( ) -> list[DataEdge]: """Join producer endpoints to consumer endpoints via the two-tier match. - The shared core behind the discovery-graph data-edge derivation (and the - IR-facing derivation added in the convert->package work): it owns the - :func:`_match_assets` tier logic, the no-self-edge rule, and the dedup, so - every phase joins identically. + The shared core behind :func:`build_data_edges` (IR) and the discovery-graph + data-edge derivation: it owns the :func:`_match_assets` tier logic, the + no-self-edge rule, and the dedup, so both phases join identically. Args: producers: ``(task_key, written asset)`` pairs, in first-seen order. @@ -138,3 +221,93 @@ def data_edges_from_endpoints( ) ) return edges + + +def build_data_edges(pipeline: Pipeline) -> list[DataEdge]: + """Derive proven producer -> consumer data hand-offs for a pipeline. + + A producer is any activity with a ``data_writes`` asset; a consumer any + activity with a ``data_reads`` asset, gathered across the whole pipeline + (ForEach / If / Switch bodies included). Each producer asset is joined against + each consumer asset via :func:`_match_assets`, tagging the edge as an + ``identity`` or ``signature`` match. An activity never hands off to itself + (no self-edges), and identical edges are collapsed (no duplicates). + + Args: + pipeline: The translated pipeline IR. + + Returns: + Deduplicated list of data edges, in first-seen order. + """ + activities = list(walk_activities(pipeline.tasks)) + producers = [(activity.task_key, asset) for activity in activities for asset in activity.data_writes] + consumers = [(activity.task_key, asset) for activity in activities for asset in activity.data_reads] + return data_edges_from_endpoints(producers, consumers) + + +def build_motif_annotations(pipeline: Pipeline) -> list[MotifAnnotation]: + """Derive motif annotations from the collapsed motif activities in a pipeline. + + One annotation per :class:`MotifActivity`, listing the task keys it spans + (the motif task itself plus any member activities that carry the same + ``motif_id`` tag). Deduplicated by ``motif_id`` in first-seen order. + + Args: + pipeline: The translated pipeline IR. + + Returns: + List of motif annotations. + """ + annotations: list[MotifAnnotation] = [] + seen: set[str] = set() + activities = list(walk_activities(pipeline.tasks)) + for activity in activities: + if not isinstance(activity, MotifActivity): + continue + if activity.motif_id in seen: + continue + seen.add(activity.motif_id) + members = [activity.task_key] + members.extend( + other.task_key for other in activities if other is not activity and other.motif_id == activity.motif_id + ) + annotations.append( + MotifAnnotation( + motif_id=activity.motif_id, + member_task_keys=members, + display_name=activity.display_name, + databricks_replacement=activity.databricks_replacement, + notes=list(activity.confidence_notes), + source_type_hint=activity.source_type_hint, + ) + ) + return annotations + + +def build_lineage(pipeline: Pipeline) -> Lineage: + """Compose the full source-neutral lineage block for a pipeline. + + Args: + pipeline: The translated pipeline IR. + + Returns: + A :class:`Lineage` with control edges, data edges, and motif annotations. + """ + return Lineage( + control_edges=build_control_edges(pipeline), + data_edges=build_data_edges(pipeline), + motifs=build_motif_annotations(pipeline), + ) + + +def with_lineage(pipeline: Pipeline, lineage: Lineage) -> Pipeline: + """Return a *new* pipeline carrying *lineage*, leaving the input untouched. + + Args: + pipeline: The pipeline to copy. + lineage: The lineage block to attach. + + Returns: + A shallow copy of *pipeline* with ``lineage`` set. + """ + return dataclasses.replace(pipeline, lineage=lineage) diff --git a/src/flowx/models/ir.py b/src/flowx/models/ir.py index 69e3eae..6c98f63 100644 --- a/src/flowx/models/ir.py +++ b/src/flowx/models/ir.py @@ -160,7 +160,8 @@ class MotifAnnotation: Source-neutral record tying a motif id to the tasks that belong to it, so the lineage block can report motifs without depending on how any particular - source detects them. ``member_task_keys`` lists the tasks the motif spans. + source detects them. ``member_task_keys`` lists the tasks the motif spans, + and each member activity also carries the same :attr:`Activity.motif_id` tag. Attributes: motif_id: Identifier of the matched motif definition. @@ -237,6 +238,13 @@ class Activity: compute_mode: str | None = None # Collapsed activity_and_notify spec set by the adapter: {destination, events, args, destination_name}. notifications: dict[str, Any] | None = None + # Lineage substrate (#61): source-neutral data assets this activity reads from and writes to, + # populated by per-source extractors in follow-up work. Always lists, never None. + data_reads: list[DataAsset] = field(default_factory=list) + data_writes: list[DataAsset] = field(default_factory=list) + # Id of the motif this activity was folded into (or belongs to); None when it is part of no motif. + # Owned here so every activity type -- not only MotifActivity -- can carry the tag. + motif_id: str | None = None @dataclass(slots=True, kw_only=True) @@ -713,6 +721,10 @@ class PlaceholderActivity(Activity): class MotifActivity(Activity): """Activity produced by collapsing a detected motif pattern. + Redeclares :attr:`Activity.motif_id` as required (the base owns the field so + every activity type can carry the tag and it round-trips through one code + path, but a motif activity always has one). + Attributes: motif_id: Identifier of the matched motif definition. display_name: Human-readable motif name. @@ -767,6 +779,8 @@ class Pipeline: reconciliation_status: Source-audit result for this pipeline. migration_status: Whether the pipeline is included or explicitly excluded. audit: Source-audit counts and transformation ledger. + lineage: Source-neutral lineage block (control/data edges + motif + annotations), or ``None`` when lineage has not been derived. """ name: str @@ -783,6 +797,7 @@ class Pipeline: audit: dict[str, Any] = field(default_factory=dict) translation_configuration: TranslationConfiguration | None = None bundle_variables: dict[str, dict[str, Any]] = field(default_factory=dict) + lineage: Lineage | None = None @dataclass(frozen=True, slots=True) diff --git a/src/flowx/sources/adf/dataset_lineage.py b/src/flowx/sources/adf/dataset_lineage.py new file mode 100644 index 0000000..46fc44a --- /dev/null +++ b/src/flowx/sources/adf/dataset_lineage.py @@ -0,0 +1,740 @@ +"""Resolve ADF dataset references into source-neutral :class:`DataAsset` values. + +This is the ADF half of data-lineage population (#62b). It re-homes the +dataset-identity resolver and the path-signature logic first written for the +closed #36 ADF-only lineage attempt, but emits the shared #61 two-tier +:class:`~flowx.models.ir.DataAsset` (``identity`` + ``signature``) instead of a +bespoke edge type, so the source-neutral join in :mod:`flowx.lineage` / +:mod:`flowx.discovery_lineage` does the matching. + +Two tiers, exactly as #36 established them: + +* **identity** -- the resolved physical location of the asset (a store-qualified + ``/schema.table`` or a concrete ``abfss://`` path down to the literal + file name). Present only when it resolves *deterministically* + from literals; a parameterised reference is never guessed at and leaves + ``identity`` unset. This is the strong join key. +* **signature** -- the *path-derived* weak key. For an asset with a resolved + identity the signature mirrors that identity, so the weak tier can never join a + resolved asset to an unresolved one on a coincidence. For an unresolved reference + the signature is the *structural path signature* (#36's "expression" tier: the + literal path skeleton plus its parameter-slot count) when the path has a literal + anchor, prefixed ``ST[]/`` when the dataset's store is known (see + :func:`_resolve_dataset_store`), so two datasets on different stores that merely + share a folder shape never join. An unknown store is never guessed: the signature + stays unqualified, and a qualified and an unqualified signature never join. + It is **never** the bare dataset reference name: two unrelated opaque + references that merely share a name must not join (#36's explicit rule), so when + neither a physical identity nor a path-anchored signature is available the + signature is left empty and the asset cannot participate in signature matching. + The signature is also left empty when the activity overrides the dataset's + location at run time (a source query, stored procedure or ``storeSettings`` path + override on a read; a sink stored procedure on a write), because the dataset's + path is then not what the activity touches. + +Only literal, provable values ever become an ``identity`` -- the resolver returns +``None`` rather than guessing, which is what stopped #36's spurious edges. +""" + +from __future__ import annotations + +import json +import re +from collections.abc import Callable, Iterator +from typing import Any + +from flowx.models.adf_ast import AdfActivity, AdfDatasetReference, AdfDefinitions +from flowx.models.ir import DataAsset, TranslationContext +from flowx.parser.expression_parser import resolve_expression, resolve_interpolated_string + +_ACCOUNT_NAME_RE = re.compile(r"AccountName=([A-Za-z0-9]+)", re.IGNORECASE) +_DATASET_PARAM_RE = re.compile(r"^@dataset\(\)\.([A-Za-z_][A-Za-z0-9_]*)$") +_ARM_EXPRESSION_RE = re.compile(r"^\[\s*[A-Za-z_][A-Za-z0-9_]*\s*\(.*\]$", re.DOTALL) +_RUN_TIME_PATH_OVERRIDES = ("wildcardFolderPath", "wildcardFileName", "fileListPath", "prefix") + +# Runtime references inside a path expression. Each is a value only knowable at +# run time; for a *structural* signature we collapse them all to one slot token so +# that, e.g., a writer's ``pipeline().parameters.entityID`` and a reader's +# ``item().entityID`` (the same value passed down a ForEach) share the same shape. +_PARAM_REF_RE = re.compile( + r"pipeline\(\)\.parameters\.\w+" + r"|item\(\)(?:\.\w+)*" + r"|variables\('[^']*'\)" + r"|dataset\(\)\.\w+" + r"|activity\('[^']*'\)\.[\w.]+" +) +_QUOTED_LITERAL_RE = re.compile(r"'([^']*)'") + + +# --------------------------------------------------------------------------- # +# Public entry point +# --------------------------------------------------------------------------- # + + +def activity_data_assets( + activity: AdfActivity, + definitions: AdfDefinitions, + context: TranslationContext | None = None, +) -> tuple[list[DataAsset], list[DataAsset]]: + """Resolve the assets an ADF activity reads and writes. + + Captures **every** input and output, not just index 0: a Copy reads its + ``source`` dataset(s) and writes its ``sink`` dataset(s), a Lookup reads its + ``dataset``, and any activity-level ``inputs`` / ``outputs`` slots are all + included. A dataset named in both an activity-level slot and ``typeProperties`` + is counted once per side so an activity does not emit two identical assets. + + When the activity overrides the dataset's physical source or target at run time, + the dataset's own location is not what the activity touches, so that side gets + neither an identity nor a signature and cannot join on either tier. Reads are overridden by + a source query or stored procedure, or by ``storeSettings`` that give a wildcard + folder or file name, a file list, or a prefix; writes by a sink stored procedure, + which decides the table itself. + + Args: + activity: The ADF activity to resolve. + definitions: All loaded ADF definitions (datasets + linked services), + needed to resolve a reference to its physical identity. + context: Optional translation context for expression resolution; a default + (empty) context is used when none is given, mirroring the deterministic + discover-time resolution the closed #36 attempt used. + + Returns: + ``(data_reads, data_writes)`` as lists of :class:`DataAsset`. + """ + resolution_context = context if context is not None else TranslationContext() + reads_overridden = _location_overridden(activity, produced=False) + reads = [ + _dataset_ref_to_asset(reference, definitions, resolution_context, location_overridden=reads_overridden) + for reference in _activity_dataset_refs(activity, produced=False) + ] + writes_overridden = _location_overridden(activity, produced=True) + writes = [ + _dataset_ref_to_asset(reference, definitions, resolution_context, location_overridden=writes_overridden) + for reference in _activity_dataset_refs(activity, produced=True) + ] + return reads, writes + + +def _location_overridden(activity: AdfActivity, *, produced: bool) -> bool: + """Say whether the activity replaces its datasets' physical target (produced) or source (not) at run time. + + A Copy sink that names a stored procedure writes wherever the procedure decides. + A Copy or Lookup ``source`` that names a query or stored procedure reads what it + returns. Connectors name these keys differently (``sqlReaderQuery``, + ``oracleReaderQuery``, ``sqlReaderStoredProcedureName``, ...), so they are + matched by kind: ``query`` or any key ending in ``Query`` or + ``StoredProcedureName``. ``storeSettings`` with a wildcard, file list or prefix + replace the read path: Copy and Lookup carry them on ``source``, Delete and + GetMetadata directly on ``typeProperties``. + """ + type_properties = activity.type_properties or {} + if produced: + return _names_any(type_properties.get("sink"), lambda key: key.endswith("StoredProcedureName")) + source = type_properties.get("source") + store_settings_candidates = ( + source.get("storeSettings") if isinstance(source, dict) else None, + type_properties.get("storeSettings"), + ) + return _names_any(source, _is_source_query_key) or any( + _names_any(store_settings, lambda key: key in _RUN_TIME_PATH_OVERRIDES) + for store_settings in store_settings_candidates + ) + + +def _is_source_query_key(key: str) -> bool: + """``True`` for a ``source`` key that names a query or stored procedure to read through.""" + return key == "query" or key.endswith("Query") or key.endswith("StoredProcedureName") + + +def _names_any(settings: object, is_override_key: Callable[[str], bool]) -> bool: + """``True`` when *settings* is a dict with a non-empty value under a key *is_override_key* accepts.""" + return isinstance(settings, dict) and any(value and is_override_key(key) for key, value in settings.items()) + + +def _dataset_ref_to_asset( + dataset_ref: AdfDatasetReference, + definitions: AdfDefinitions, + context: TranslationContext, + *, + location_overridden: bool, +) -> DataAsset: + """Turn one dataset reference into a two-tier :class:`DataAsset`. + + Signature is derived only from the resolved *physical* location -- the identity + when it resolves, else the structural path signature. It is **never** the bare + dataset reference name (#36's hard rule): two unrelated opaque references that + merely share a name must not join, so when neither a physical identity nor a + path-anchored signature is available the signature is left empty. An empty + signature is falsy, so :func:`~flowx.lineage._match_assets` cannot use it as a + join key -- the asset is still captured as a read / write for reporting, it just + cannot manufacture a signature-tier edge. When *location_overridden* is set the + dataset's location is not what the activity touches, so neither an identity nor a + signature is derived from it. + + A path signature is prefixed with the dataset's store when that store is known + (see :func:`_resolve_dataset_store`), so two datasets on different containers or + accounts that merely share a folder shape never join. + """ + if location_overridden: + return DataAsset(signature="", identity=None, asset_type=_asset_type(dataset_ref, definitions)) + identity = resolve_dataset_identity(dataset_ref, definitions, context) + if identity is not None: + # Mirror the identity into the signature so the weak tier never joins a + # resolved asset to an unresolved one that merely shares a physical value. + return DataAsset(signature=identity, identity=identity, asset_type=_asset_type(dataset_ref, definitions)) + path_signature = _path_signature(dataset_ref.parameters) + store = _resolve_dataset_store(dataset_ref, definitions, context) if path_signature is not None else None + if path_signature is not None and store: + path_signature = f"ST[{store}]/{path_signature}" + return DataAsset( + signature=path_signature if path_signature is not None else "", + identity=None, + asset_type=_asset_type(dataset_ref, definitions), + ) + + +# --------------------------------------------------------------------------- # +# Dataset reference gathering (all inputs / outputs) +# --------------------------------------------------------------------------- # + + +def _typeprops_dataset_ref(candidate: object) -> AdfDatasetReference | None: + """Build a dataset reference from a ``typeProperties`` source/sink/dataset slot. + + Carries the slot's ``parameters`` (Lookup / Delete / GetMetadata put the + dataset call-site params here) so the path signature can be computed. + """ + if isinstance(candidate, dict): + name = candidate.get("referenceName") + if isinstance(name, str) and name: + parameters = candidate.get("parameters") + return AdfDatasetReference( + reference_name=name, + parameters=parameters if isinstance(parameters, dict) else None, + ) + return None + + +def _activity_dataset_refs(activity: AdfActivity, *, produced: bool) -> Iterator[AdfDatasetReference]: + """Yield the dataset references an activity writes (produced) or reads (not). + + Gathers activity-level ``inputs`` / ``outputs`` **and** the ``typeProperties`` + ``source`` / ``sink`` / ``dataset`` slots. De-duplication is by + ``(reference_name, parameter binding)``, not by name alone: a dataset named in + both an activity slot and ``typeProperties`` with the *same* call-site params is + the same physical asset and is yielded once, but two uses of the *same* + parameterised dataset with *different* params (``ds(tbl=orders)`` vs + ``ds(tbl=customers)``) resolve to distinct physical assets and are both kept. + """ + type_properties = activity.type_properties or {} + candidates: list[AdfDatasetReference] = [] + if produced: + candidates.extend(activity.outputs or []) + sink_reference = _typeprops_dataset_ref(type_properties.get("sink")) + if sink_reference is not None: + candidates.append(sink_reference) + else: + candidates.extend(activity.inputs or []) + for key in ("source", "dataset"): + read_reference = _typeprops_dataset_ref(type_properties.get(key)) + if read_reference is not None: + candidates.append(read_reference) + + seen: set[tuple[str, str]] = set() + for reference in candidates: + dedupe_key = (reference.reference_name, _parameter_binding_key(reference)) + if dedupe_key in seen: + continue + seen.add(dedupe_key) + yield reference + + +def _parameter_binding_key(reference: AdfDatasetReference) -> str: + """Stable key for a reference's call-site parameter binding. + + Two references with the same name collapse only when their parameters match, so + distinct bindings that resolve to distinct physical assets survive. Sorted keys + make the string order-independent; ``default=str`` keeps it total for any value + an ADF export can carry. + """ + return json.dumps(reference.parameters or {}, sort_keys=True, default=str) + + +# --------------------------------------------------------------------------- # +# Physical identity resolution (tier 1) +# --------------------------------------------------------------------------- # + + +def resolve_dataset_identity( + dataset_ref: AdfDatasetReference, + definitions: AdfDefinitions, + context: TranslationContext | None = None, +) -> str | None: + """Deterministic physical identity for a dataset reference. + + Returns ``"/schema.table"`` when a table is resolvable, else a storage + path down to the literal file name, else ``None`` (never a guess). Used to join + producers to consumers on the same physical asset even when their ADF dataset + names differ. + + A table name is only unique within the store that holds it, so the table is + prefixed with the backing linked service's server (and database, when given), + or with the linked service name when the server is hidden in a secret. + Otherwise a source ``dbo.Orders`` and a warehouse ``dbo.Orders`` would look like + one table. + + Parameterised values (ADF expressions or DAB-ref placeholders), including a + parameterised linked service, are treated as unresolvable and return ``None`` + -- they must never be used as identity keys because two unrelated pipelines + sharing a parameter name would collide on the same placeholder string. + """ + resolution_context = context if context is not None else TranslationContext() + properties = _dataset_props(dataset_ref, definitions) + if properties is None: + return None + schema, table = _resolve_table_reference(dataset_ref, properties, resolution_context) + if table: + if not all(_is_physical(part) for part in (schema, table) if part): + return None + qualified_table = f"{schema}.{table}" if schema else table + store = _resolve_table_store(properties, definitions) + return f"{store}/{qualified_table}" if store else None + return _resolve_dataset_path( + properties, + _backing_linked_service(properties, definitions), + _effective_dataset_params(dataset_ref, properties), + resolution_context, + ) + + +def _dataset_props(dataset_ref: AdfDatasetReference, definitions: AdfDefinitions) -> dict[str, Any] | None: + """Return the ``properties`` dict for a dataset reference, or ``None``.""" + dataset = definitions.get_dataset(dataset_ref.reference_name) + if not dataset: + return None + return dict(dataset.properties or {}) + + +def _resolve_param_value( + raw: Any, + dataset_params: dict[str, Any], + context: TranslationContext, +) -> str: + """Resolve a single ADF location / table field to a string.""" + if raw is None: + return "" + if isinstance(raw, dict) and raw.get("type") == "Expression": + raw = raw.get("value", "") + if isinstance(raw, (list, dict)): + return "" + if not isinstance(raw, str): + return str(raw) + text = raw + + match = _DATASET_PARAM_RE.match(text.strip()) + if match: + parameter_name = match.group(1) + return _resolve_param_value(dataset_params.get(parameter_name, ""), dataset_params, context) + + if "@{" in text: + return resolve_interpolated_string(text, context) + + if text.startswith("@"): + result = resolve_expression(text, context) + if result is not None and result.kind in ("literal", "dab_ref"): + return result.value + return text + + return text + + +def _effective_dataset_params(dataset_ref: AdfDatasetReference, dataset_props: dict[str, Any]) -> dict[str, Any]: + """Effective parameter map: declared dataset defaults first, call-site overrides win.""" + declared = dataset_props.get("parameters") or {} + effective: dict[str, Any] = {} + for name, spec in declared.items(): + if isinstance(spec, dict) and "defaultValue" in spec: + effective[name] = spec["defaultValue"] + if dataset_ref.parameters: + effective.update(dict(dataset_ref.parameters)) + return effective + + +def _resolve_table_reference( + dataset_ref: AdfDatasetReference, + dataset_props: dict[str, Any] | None, + context: TranslationContext, +) -> tuple[str | None, str | None]: + """Resolve ``(schema, table)`` from a dataset reference. + + Handles both the nested ``typeProperties`` shape and the + ``schemaTypePropertiesSchema`` flattened form ``az datafactory dataset show`` + emits. ADF parameter expressions resolve against the reference's effective + parameter map. + """ + if not dataset_props: + return None, None + type_props = dataset_props.get("typeProperties") if isinstance(dataset_props.get("typeProperties"), dict) else None + effective_params = _effective_dataset_params(dataset_ref, dataset_props) + schema_raw = _pick_dataset_field( + type_props, + dataset_props, + ("schema", "database"), + ("schemaTypePropertiesSchema", "database"), + ) + table_raw = _pick_dataset_field( + type_props, + dataset_props, + ("table", "tableName"), + ("table", "tableName"), + ) + schema = _resolve_param_value(schema_raw, effective_params, context) if schema_raw is not None else None + table = _resolve_param_value(table_raw, effective_params, context) if table_raw is not None else None + return (schema or None), (table or None) + + +def _pick_dataset_field( + type_props: dict[str, Any] | None, + dataset_props: dict[str, Any], + nested_keys: tuple[str, ...], + flat_keys: tuple[str, ...], +) -> Any: + """First populated dataset field across the nested and az-flattened shapes. + + Empty strings, empty lists, and ``None`` are skipped so a column-schema + artifact like ``schema: []`` does not shadow the real database schema stored + under a flattened key. + """ + candidates: list[Any] = [] + if type_props is not None: + candidates.extend(type_props.get(key) for key in nested_keys) + candidates.extend(dataset_props.get(key) for key in flat_keys) + for value in candidates: + if value is None: + continue + if isinstance(value, (list, dict)) and not value: + continue + if isinstance(value, str) and not value.strip(): + continue + return value + return None + + +def _linked_service_reference(dataset_props: dict[str, Any]) -> tuple[str, dict[str, Any]]: + """Return the dataset's linked service name and the parameters its reference binds.""" + linked_service_reference = dataset_props.get("linkedServiceName") or {} + if not isinstance(linked_service_reference, dict): + return str(linked_service_reference), {} + parameters = linked_service_reference.get("parameters") + return linked_service_reference.get("referenceName") or "", parameters if isinstance(parameters, dict) else {} + + +def _backing_linked_service(dataset_props: dict[str, Any], definitions: AdfDefinitions) -> Any: + """Return the loaded definition of the dataset's linked service, or ``None``.""" + linked_service_name, _ = _linked_service_reference(dataset_props) + return definitions.get_linked_service(linked_service_name) if linked_service_name else None + + +def _resolve_table_store(dataset_props: dict[str, Any], definitions: AdfDefinitions) -> str | None: + """Name the store that holds a dataset's table, or ``None`` when it is not provable. + + Uses the literal ``server`` (plus ``database`` when given) from the backing + linked service's properties or plaintext connection string, exactly as written. + When the server is hidden (a Key Vault reference or a masked secret) the linked + service name stands in for it. A parameterised server, database, or connection + string, or a linked service that takes parameters, gives ``None``: each binding + may point at a different store, so no single name identifies it. + """ + linked_service_name, _ = _linked_service_reference(dataset_props) + if not linked_service_name: + return None + linked_service = definitions.get_linked_service(linked_service_name) + linked_service_properties = linked_service.properties if linked_service is not None else {} + type_props = linked_service_properties.get("typeProperties") or linked_service_properties + connection_string = type_props.get("connectionString") + if isinstance(connection_string, dict): + connection_string = connection_string.get("value") + if isinstance(connection_string, str) and not _is_physical(connection_string): + return None + connection_fields = _connection_string_fields(connection_string) if isinstance(connection_string, str) else {} + + server = type_props.get("server") or connection_fields.get("server") or connection_fields.get("data source") + database = ( + type_props.get("database") or connection_fields.get("database") or connection_fields.get("initial catalog") + ) + store_parts = [part for part in (server, database) if part] + if not all(isinstance(part, str) and part.strip() and _is_physical(part) for part in store_parts): + return None + if server: + return "/".join(part.strip() for part in store_parts) + return _unparameterised_linked_service_name(dataset_props, definitions) + + +def _unparameterised_linked_service_name(dataset_props: dict[str, Any], definitions: AdfDefinitions) -> str | None: + """The dataset's linked service name when that linked service takes no parameters, else ``None``. + + A linked service with no parameters points at the same store on every use, so its + name can stand in for a server or account it hides in a secret. One that takes + parameters, on its reference or in its own definition, may point somewhere else + for each binding. + """ + linked_service_name, reference_parameters = _linked_service_reference(dataset_props) + if not linked_service_name or reference_parameters: + return None + linked_service = definitions.get_linked_service(linked_service_name) + if linked_service is not None and linked_service.properties.get("parameters"): + return None + return linked_service_name + + +def _resolve_dataset_store( + dataset_ref: AdfDatasetReference, + definitions: AdfDefinitions, + context: TranslationContext, +) -> str | None: + """Name the store a dataset lives in, even when its full path or table does not resolve. + + A table dataset's store is the one :func:`_resolve_table_store` names. A file + dataset's store is ``@`` when both are literal. When the + file system is literal but the account cannot be read (a Key Vault or masked + connection string, for example), the linked service name stands in for the + account, as long as the linked service takes no parameters. ``None`` whenever + the store is not provable, so the caller leaves the signature unqualified rather + than guess. + """ + properties = _dataset_props(dataset_ref, definitions) + if properties is None: + return None + _, table = _resolve_table_reference(dataset_ref, properties, context) + if table: + return _resolve_table_store(properties, definitions) + type_props = properties.get("typeProperties") or properties + location = type_props.get("location") + if not isinstance(location, dict): + return None + file_system, account = _resolve_file_store( + location, + _backing_linked_service(properties, definitions), + _effective_dataset_params(dataset_ref, properties), + context, + ) + if file_system is None: + return None + store = account or _unparameterised_linked_service_name(properties, definitions) + return f"{file_system}@{store}" if store else None + + +def _resolve_file_store( + location: dict[str, Any], + linked_service: Any, + dataset_params: dict[str, Any], + context: TranslationContext, +) -> tuple[str | None, str | None]: + """Resolve a file dataset's ``(file system, storage account)``. + + Each part is ``None`` unless it resolves to a literal. The path identity and the + signature's store qualifier both read the store here, so they always agree on + what counts as a known file system and account. + """ + file_system = _resolve_param_value(location.get("fileSystem") or location.get("container"), dataset_params, context) + account = _resolve_storage_account(linked_service) + return ( + file_system if file_system and _is_physical(file_system) else None, + account if account and _is_physical(account) else None, + ) + + +def _connection_string_fields(connection_string: str) -> dict[str, str]: + """Split a ``key=value;`` connection string into a lower-cased key map.""" + fields: dict[str, str] = {} + for part in connection_string.split(";"): + key, separator, value = part.partition("=") + if separator: + fields[key.strip().lower()] = value.strip() + return fields + + +def _resolve_dataset_path( + dataset_props: dict[str, Any], + linked_service: Any, + dataset_params: dict[str, Any], + context: TranslationContext, +) -> str | None: + """Resolve a dataset's storage path from its location + backing linked service. + + The path runs down to the literal ``fileName`` when the dataset names one, so + two files in the same folder stay distinct. Location parts written as + ``@dataset().x`` resolve against the reference's effective parameters, as table + names do, so a parameterised dataset bound to literals still gets its identity. + Any part that stays parameterised or unresolved (file system, folder, file name, + or storage account) yields ``None``. + """ + type_props = dataset_props.get("typeProperties") or dataset_props + location = type_props.get("location") or {} + if not isinstance(location, dict): + return None + + def _resolved_part(raw: Any) -> str | None: + if raw is None or raw == "": + return "" + resolved = _resolve_param_value(raw, dataset_params, context) + # A part that was written but resolves to nothing (an unbound parameter) is unknown, not absent. + if not resolved or not _is_physical(resolved): + return None + return resolved + + file_system, account = _resolve_file_store(location, linked_service, dataset_params, context) + folder_path = _resolved_part(location.get("folderPath")) + file_name = _resolved_part(location.get("fileName")) + if file_system is None or account is None or folder_path is None or file_name is None: + return None + + relative_path = "/".join(part.strip("/") for part in (folder_path, file_name) if part.strip("/")) + return f"abfss://{file_system}@{account}.dfs.core.windows.net/{relative_path}".rstrip("/") + + +def _resolve_storage_account(linked_service: Any) -> str | None: + """Pull a storage account name out of a linked service, if present. + + The whole ``url``, ``sasUri`` or connection string must be physical before the + account is cut out of it: cutting first would drop the closing ``]`` of an ARM + expression or the tail of an ``@{...}`` and leave a fragment that looks literal. + """ + if linked_service is None: + return None + type_props = linked_service.properties.get("typeProperties") or linked_service.properties + + url = type_props.get("url") or "" + if isinstance(url, str) and url: + if not _is_physical(url): + return None + host = url.replace("https://", "").split("/", 1)[0] + host_no_port = host.split(":", 1)[0] + if "." in host_no_port: + return host_no_port.split(".", 1)[0] + + sas_uri = type_props.get("sasUri") or "" + if isinstance(sas_uri, str) and sas_uri: + if not _is_physical(sas_uri): + return None + host = sas_uri.split("?", 1)[0].replace("https://", "").split("/", 1)[0] + if "." in host: + return host.split(".", 1)[0] + + # Plaintext connection string (rare in az exports -- usually masked). + connection_string = type_props.get("connectionString") + if isinstance(connection_string, dict): + connection_string = connection_string.get("value", "") + if isinstance(connection_string, str): + if not _is_physical(connection_string): + return None + match = _ACCOUNT_NAME_RE.search(connection_string) + if match: + return match.group(1) + + # AWS -- bucket name lives on the dataset, account is implicit; nothing useful + # to return at the linked-service level for S3 / GCS. + return None + + +def _is_physical(value: str) -> bool: + """Return ``True`` only when *value* is a literal (physical) identifier. + + A value is NOT physical when it still contains an unresolved marker -- a + DAB-ref placeholder (``{{`` ... ``}}``), a leftover ADF interpolation + fragment (``@{``), a bare ADF expression (starts with ``@``), or an ARM + template expression the export left unevaluated (``[parameters('x')]``, + ``[concat(...)]``). An ARM expression always opens with a function call, so + bracket-quoted SQL names such as ``[dbo].[Orders]`` stay physical, and so + does ARM's ``[[`` escape for a literal that starts with ``[``. + """ + stripped = value.strip() + if "{{" in value or "@{" in value or stripped.startswith("@"): + return False + return not _ARM_EXPRESSION_RE.match(stripped) + + +# --------------------------------------------------------------------------- # +# Structural path signature (tier 2, #36's "expression" tier) +# --------------------------------------------------------------------------- # + + +def _path_signature(parameters: dict[str, Any] | None) -> str | None: + """Structural signature of a reference's parameterised folderPath / fileName. + + Requires a resolvable **folderPath** literal anchor. A file name alone is too + weak a discriminator: many unrelated activities write ``.csv`` / ``.json`` files + to opaque parameterised folders, so a signature built only from a file extension + would join them all -- re-creating the explosion the identity-only join avoids. + Anchoring on the literal folder segment keeps the match specific to a real, + named location. + + ``None`` when the folder path has no literal segment to anchor on. + """ + if not parameters: + return None + folder_signature = _normalize_path_expression(parameters.get("folderPath")) + if folder_signature is None: + return None + file_signature = _normalize_path_expression(parameters.get("fileName")) + return f"FP[{folder_signature}]/FN[{file_signature}]" + + +def _normalize_path_expression(expression: Any) -> str | None: + """Reduce a (possibly parameterised) ADF path expression to a structural signature. + + Keeps the literal path segments and collapses every runtime reference to a + single ``

`` slot, so the result captures the path *shape* (literal skeleton + plus slot count) without guessing the runtime value. Returns ``None`` when + there is no literal segment to anchor on (a signature of only slots is too weak + a join key -- never guess). + """ + if isinstance(expression, dict): + expression = expression.get("value", "") + if not isinstance(expression, str) or not expression.strip(): + return None + text = expression.strip() + if "@" not in text: + # A bare literal value (no ADF expression): the whole string is the literal + # path / filename, with no runtime slots. + literal = re.sub(r"/+", "/", text).strip("/") + return f"{literal}|slots=0" if literal else None + marked = _PARAM_REF_RE.sub("

", text) + literal = re.sub(r"/+", "/", "".join(_QUOTED_LITERAL_RE.findall(marked))).strip("/") + if not literal: + return None + return f"{literal}|slots={marked.count('

')}" + + +# --------------------------------------------------------------------------- # +# Asset-type classification (best-effort neutral kind) +# --------------------------------------------------------------------------- # + +_TABLE_HINTS = ("table", "sql", "database") +_FILE_HINTS = ("delimited", "parquet", "orc", "avro", "json", "binary", "excel", "xml", "blob", "adls", "file") + + +def _asset_type(dataset_ref: AdfDatasetReference, definitions: AdfDefinitions) -> str | None: + """Best-effort neutral asset kind (``"table"`` / ``"file"``), or ``None``. + + Prefers the shape of the dataset's typeProperties (a ``location`` block means a + file, a ``table`` / ``schema`` means a table) and falls back to keyword hints in + the dataset's ADF type string. Returns ``None`` when nothing is conclusive + rather than guessing. + """ + dataset = definitions.get_dataset(dataset_ref.reference_name) + if dataset is None: + return None + type_props = dataset.properties.get("typeProperties") + if isinstance(type_props, dict): + if isinstance(type_props.get("location"), dict): + return "file" + if type_props.get("table") or type_props.get("tableName") or type_props.get("schema"): + return "table" + dataset_type = (dataset.type or "").lower() + if any(hint in dataset_type for hint in _TABLE_HINTS): + return "table" + if any(hint in dataset_type for hint in _FILE_HINTS): + return "file" + return None diff --git a/src/flowx/sources/adf/discovery_mapping.py b/src/flowx/sources/adf/discovery_mapping.py new file mode 100644 index 0000000..dee2c99 --- /dev/null +++ b/src/flowx/sources/adf/discovery_mapping.py @@ -0,0 +1,406 @@ +"""Map the ADF AST onto the source-neutral discovery graph contract (``SourceGraph``). + +This is the ADF half of the discovery contract (issue #62): it turns the typed +ADF AST (:mod:`flowx.models.adf_ast`) into the shared +:class:`~flowx.models.discovery.SourceGraph` model that both ADF and Airflow +align to. The mapping is deliberately **1:1 and lossless**: + +* every ADF activity becomes exactly one discovery node -- no motif collapse, + no merging; +* the activity's own type string is kept verbatim as ``native_type``; +* the verbatim source dict is preserved on every node (``raw``) and on the graph; +* dependency edges keep **all** of their outcome conditions (a ``dependsOn`` with + ``["Succeeded", "Skipped"]`` stays a two-condition edge); +* anything without a shared typed home -- the ADF folder, retry policy detail, + the target-side translation strategy -- rides in the ``properties`` / + ``extensions`` seams rather than being dropped or forced into a field. + +Data lineage (#62b) is populated here now: every node's ``data_reads`` / +``data_writes`` are resolved from the ADF dataset references via +:func:`~flowx.sources.adf.dataset_lineage.activity_data_assets` (the two-tier +identity / signature model), and every ``ExecutePipeline`` node records the child +pipeline it invokes under the neutral +:data:`~flowx.discovery_lineage.INVOKES_WORKFLOW_PROPERTY` marker. Because +:func:`_activity_to_node` recurses into every control-flow branch, this population +is Switch-aware for free -- a Copy or ExecutePipeline nested inside a Switch case +(or default), a ForEach / Until body, or either If branch carries its assets and +marker like any top-level node. The graph's :attr:`SourceGraph.lineage` block is +then derived by the source-neutral :func:`~flowx.discovery_lineage.build_graph_lineage`. + +The control-flow container shape follows :class:`~flowx.models.discovery.ContainerNode`: +an ``IfCondition`` becomes ``{"true": [...], "false": [...]}``, a ``ForEach`` / +``Until`` becomes ``{"body": [...]}``, and a ``Switch`` becomes +``{"": [...], "default": [...]}``. Branch insertion order matches the +order ADF declares the branches so a downstream flatten reproduces source order. +A case whose value is literally ``"default"`` is keyed ``"case:default"`` (with extra +``case:`` prefixes if another case already uses that value) so it never collides with +the Switch's own default branch or another case. +""" + +from __future__ import annotations + +from typing import Any + +from flowx.discovery_inventory import STRATEGY_PROPERTY +from flowx.discovery_lineage import INVOKES_WAIT_PROPERTY, INVOKES_WORKFLOW_PROPERTY, build_graph_lineage +from flowx.models.adf_ast import AdfActivity, AdfDefinitions, AdfParameter, AdfPipeline, AdfTrigger, AdfVariable +from flowx.models.discovery import ( + CONCEPT_BRANCH, + CONCEPT_COPY_DATA, + CONCEPT_GAP, + CONCEPT_LOOP, + CONCEPT_NOTEBOOK, + CONCEPT_QUERY, + CONCEPT_RUN_WORKFLOW, + CONCEPT_SCRIPT, + CONCEPT_SET_VARIABLE, + CONCEPT_SWITCH, + CONCEPT_WAIT, + SOURCE_ADF, + ContainerNode, + GapNode, + ParameterSpec, + PolicySpec, + ScheduleSpec, + SourceDependency, + SourceGraph, + SourceNode, +) +from flowx.sources.adf.dataset_lineage import activity_data_assets +from flowx.sources.adf.loader import classify_activity + +# Neutral trigger category for each ADF trigger type. Anything unrecognised maps +# to ``""`` (unknown) with the ADF type preserved verbatim in the schedule +# extensions, so no trigger information is lost even for a type not listed here. +_TRIGGER_KIND: dict[str, str] = { + "ScheduleTrigger": "schedule", + "TumblingWindowTrigger": "interval", + "BlobEventsTrigger": "file_arrival", + "CustomEventsTrigger": "event", +} + +# Neutral concept for each ADF activity type. The discovery concept vocabulary is +# intentionally non-exhaustive: ADF types with no shared concept (WebActivity, +# Delete, Filter, and the agentic-only types) map to CONCEPT_GAP, which here means +# "no shared concept applies" -- independent of the translation strategy, which is +# recorded separately under properties[STRATEGY_PROPERTY]. +_CONCEPT_BY_TYPE: dict[str, str] = { + "Copy": CONCEPT_COPY_DATA, + "DatabricksNotebook": CONCEPT_NOTEBOOK, + "DatabricksSparkJar": CONCEPT_SCRIPT, + "DatabricksSparkPython": CONCEPT_SCRIPT, + "DatabricksJob": CONCEPT_RUN_WORKFLOW, + "ExecutePipeline": CONCEPT_RUN_WORKFLOW, + "ForEach": CONCEPT_LOOP, + "Until": CONCEPT_LOOP, + "IfCondition": CONCEPT_BRANCH, + "Switch": CONCEPT_SWITCH, + "SetVariable": CONCEPT_SET_VARIABLE, + "AppendVariable": CONCEPT_SET_VARIABLE, + "Wait": CONCEPT_WAIT, + "Lookup": CONCEPT_QUERY, +} + + +def adf_definitions_to_source_graphs(definitions: AdfDefinitions) -> list[SourceGraph]: + """Map every pipeline in *definitions* to a shared :class:`SourceGraph`. + + Pipeline order is preserved so downstream consumers see the same ordering the + loader produced. ADF triggers are mapped into :class:`ScheduleSpec` and attached + to each pipeline they reference (see :func:`_attach_schedules`), so schedule / + trigger information is preserved rather than dropped. Each graph's + :attr:`SourceGraph.lineage` block is derived (control + data edges) once its + nodes' reads / writes and invocation markers are populated -- data edges are + joined within a single graph, matching the per-workflow scope of the shared + :class:`~flowx.models.ir.Lineage` model. + + ``definitions`` is threaded down to the activity mapper so a node's data assets + resolve against the factory's datasets and linked services. + """ + graphs = [adf_pipeline_to_source_graph(pipeline, definitions) for pipeline in definitions.pipelines] + _attach_schedules(graphs, definitions.triggers) + for graph in graphs: + graph.lineage = build_graph_lineage(graph) + return graphs + + +def _attach_schedules(graphs: list[SourceGraph], triggers: list[AdfTrigger]) -> None: + """Populate each graph's ``schedule`` from the triggers that reference it. + + A trigger can drive several pipelines and a pipeline can be driven by several + triggers. The first trigger to reference a pipeline becomes its typed + :attr:`SourceGraph.schedule`; any further triggers for the same pipeline are + preserved verbatim under ``schedule.extensions["additional_triggers"]`` so + nothing is lost. Pipeline references are matched case-insensitively, mirroring + ADF's case-insensitive identifier semantics. + + A **fresh** :class:`ScheduleSpec` is built for each pipeline assignment rather + than sharing one instance across every pipeline a trigger references: a shared + instance would let a later trigger's mutation (appending to + ``additional_triggers``) leak onto every other pipeline that trigger touched. + """ + graphs_by_name = {graph.name: graph for graph in graphs} + graphs_by_lower = {graph.name.lower(): graph for graph in graphs} + + for trigger in triggers: + for pipeline_name in _trigger_pipeline_names(trigger): + graph = graphs_by_name.get(pipeline_name) or graphs_by_lower.get(pipeline_name.lower()) + if graph is None: + continue + if graph.schedule is None: + # Per-pipeline instance: no shared mutable state across pipelines. + graph.schedule = _trigger_to_schedule(trigger) + else: + additional = graph.schedule.extensions.setdefault("additional_triggers", []) + additional.append( + { + "trigger_name": trigger.name, + "trigger_type": trigger.type, + "properties": trigger.properties, + } + ) + + +def _trigger_to_schedule(trigger: AdfTrigger) -> ScheduleSpec: + """Map a single ADF trigger to a source-faithful :class:`ScheduleSpec`. + + The recurrence payload is kept as-given: schedule triggers nest it under + ``typeProperties.recurrence`` while tumbling-window / event triggers put their + detail directly in ``typeProperties``, so whichever is present becomes the + verbatim ``expression``. The full trigger ``properties`` block also rides in + ``extensions`` so the mapping is lossless even for trigger detail with no typed + home yet (pipeline parameters, runtime state, annotations). + """ + type_properties = trigger.properties.get("typeProperties") or {} + recurrence = type_properties.get("recurrence") if isinstance(type_properties, dict) else None + if isinstance(recurrence, dict): + expression: Any = recurrence + timezone = recurrence.get("timeZone") + else: + expression = type_properties or None + timezone = type_properties.get("timeZone") if isinstance(type_properties, dict) else None + + return ScheduleSpec( + kind=_TRIGGER_KIND.get(trigger.type, ""), + expression=expression, + timezone=timezone, + extensions={ + "trigger_name": trigger.name, + "trigger_type": trigger.type, + "properties": trigger.properties, + }, + ) + + +def _trigger_pipeline_names(trigger: AdfTrigger) -> list[str]: + """Collect the names of the pipelines a trigger references, in order.""" + names: list[str] = [] + for reference in trigger.pipelines or []: + pipeline_reference = reference.get("pipelineReference") if isinstance(reference, dict) else None + name = pipeline_reference.get("referenceName") if isinstance(pipeline_reference, dict) else None + if name: + names.append(name) + return names + + +def adf_pipeline_to_source_graph(pipeline: AdfPipeline, definitions: AdfDefinitions | None = None) -> SourceGraph: + """Map a single ADF pipeline to a source-faithful :class:`SourceGraph`. + + ``definitions`` supplies the datasets and linked services the activity mapper + needs to resolve each node's data assets to a physical identity. It is optional + so a caller mapping a pipeline in isolation still works; without it, dataset + references simply stay unresolved (identity ``None``) and fall back to their + neutral signature. It does **not** attach the graph-level lineage block -- + :func:`adf_definitions_to_source_graphs` owns that, once every graph is built. + """ + resolved_definitions = definitions if definitions is not None else AdfDefinitions(pipelines=[]) + properties: dict[str, str] = {} + if pipeline.folder: + properties["folder"] = pipeline.folder + + return SourceGraph( + name=pipeline.name, + source=SOURCE_ADF, + parameters=_parameters_to_specs(pipeline.parameters), + variables=_variables_to_specs(pipeline.variables), + tags=list(pipeline.annotations) if pipeline.annotations else [], + tasks=[_activity_to_node(activity, resolved_definitions) for activity in pipeline.activities], + properties=properties, + raw=pipeline.raw, + ) + + +def _activity_to_node(activity: AdfActivity, definitions: AdfDefinitions) -> SourceNode: + """Map one ADF activity to a discovery node (1:1, source-faithful). + + Fields are passed explicitly to each node class (rather than unpacking a + shared dict) so the mapping stays type-checked end to end. Data reads / writes + are resolved from the activity's dataset references, and an ``ExecutePipeline`` + records the child pipeline it invokes under the neutral control-edge marker; + both apply to nested nodes too because the branch children below route back + through this function. + """ + strategy = classify_activity(activity.type) + concept = _CONCEPT_BY_TYPE.get(activity.type, CONCEPT_GAP) + dependencies = [ + SourceDependency(upstream=dependency.activity, conditions=list(dependency.dependency_conditions)) + for dependency in (activity.depends_on or []) + ] + policy = _policy_to_spec(activity) + properties: dict[str, Any] = {STRATEGY_PROPERTY: strategy.value} + _record_invocation(activity, properties) + data_reads, data_writes = activity_data_assets(activity, definitions) + + branches = _control_flow_branches(activity, definitions) + if branches is not None: + return ContainerNode( + source_id=activity.name, + task_key=activity.name, + concept=concept, + source=SOURCE_ADF, + name=activity.name, + native_type=activity.type, + dependencies=dependencies, + policy=policy, + data_reads=data_reads, + data_writes=data_writes, + properties=properties, + raw=activity.raw, + branches=branches, + ) + if strategy.value == "unsupported": + return GapNode( + source_id=activity.name, + task_key=activity.name, + source=SOURCE_ADF, + name=activity.name, + native_type=activity.type, + dependencies=dependencies, + policy=policy, + data_reads=data_reads, + data_writes=data_writes, + properties=properties, + raw=activity.raw, + reason=f"unsupported ADF activity type {activity.type!r}", + ) + return SourceNode( + source_id=activity.name, + task_key=activity.name, + concept=concept, + source=SOURCE_ADF, + name=activity.name, + native_type=activity.type, + dependencies=dependencies, + policy=policy, + data_reads=data_reads, + data_writes=data_writes, + properties=properties, + raw=activity.raw, + ) + + +def _record_invocation(activity: AdfActivity, properties: dict[str, Any]) -> None: + """Stash the child pipeline an ``ExecutePipeline`` invokes under neutral keys. + + The callee reference name and ``waitOnCompletion`` flag are read from the same + ADF ``typeProperties`` the convert-time translator reads, then recorded under + the source-neutral :data:`INVOKES_WORKFLOW_PROPERTY` / :data:`INVOKES_WAIT_PROPERTY` + keys so the source-agnostic control-edge derivation can find them without + knowing anything about ADF. An empty / missing callee still records the marker + (with an empty target) so the edge is reported as unresolved rather than dropped. + """ + if activity.type != "ExecutePipeline": + return + type_properties = activity.type_properties or {} + reference = type_properties.get("pipeline", {}) + if isinstance(reference, dict): + callee = reference.get("referenceName", "") or "" + else: + callee = str(reference) + properties[INVOKES_WORKFLOW_PROPERTY] = callee + properties[INVOKES_WAIT_PROPERTY] = bool(type_properties.get("waitOnCompletion", True)) + + +def _control_flow_branches(activity: AdfActivity, definitions: AdfDefinitions) -> dict[str, list[SourceNode]] | None: + """Return the labelled branches of a control-flow activity, or ``None`` if it is a leaf. + + Keyed on the activity **type**, not on whether children happen to be present: + a control-flow activity always maps to a :class:`ContainerNode`, and every + branch it declares is always represented -- an empty branch is present-but-empty, + never omitted. So an empty ``IfCondition`` still yields ``{"true": [], "false": []}`` + and a one-sided ``If`` keeps its empty ``false`` branch, rather than collapsing to + a plain node and losing the control-flow structure. Returning ``None`` (not an + empty dict) is what tells the caller the activity is a leaf. + + Branch order matches ADF's own declaration order (true before false, cases + before default) so a downstream flatten walks children in source order. + """ + if activity.type == "IfCondition": + return { + "true": [_activity_to_node(child, definitions) for child in (activity.if_true_activities or [])], + "false": [_activity_to_node(child, definitions) for child in (activity.if_false_activities or [])], + } + if activity.type in ("ForEach", "Until"): + return {"body": [_activity_to_node(child, definitions) for child in (activity.activities or [])]} + if activity.type == "Switch": + branches: dict[str, list[SourceNode]] = {} + parsed_cases = activity.switch_cases or {} + # The parser drops a case whose activities list is empty, so the declared case labels come + # from the raw typeProperties; that keeps every named branch, empty ones included. + declared_cases = (activity.type_properties or {}).get("cases") + declared_values = [str(case.get("value", "")) for case in declared_cases or []] + default_case_label = "case:default" + while default_case_label in declared_values: + default_case_label = f"case:{default_case_label}" + for case_value in declared_values: + case_label = default_case_label if case_value == "default" else case_value + branches[case_label] = [_activity_to_node(child, definitions) for child in parsed_cases.get(case_value, [])] + branches["default"] = [ + _activity_to_node(child, definitions) for child in (activity.switch_default_activities or []) + ] + return branches + return None + + +def _policy_to_spec(activity: AdfActivity) -> PolicySpec | None: + """Map an ADF retry/timeout policy to a :class:`PolicySpec`, if present. + + Retry count and interval are already numeric in ADF and map to typed fields. + The ADF timeout is an ISO-8601 duration *string* -- normalising it to seconds + is a target concern, so it rides verbatim in ``extensions`` alongside the + ``secureInput`` / ``secureOutput`` flags rather than being guessed at here. + """ + policy = activity.policy + if policy is None: + return None + extensions: dict[str, object] = {} + if policy.timeout is not None: + extensions["timeout"] = policy.timeout + if policy.secure_input: + extensions["secure_input"] = True + if policy.secure_output: + extensions["secure_output"] = True + return PolicySpec( + max_retries=policy.retry, + retry_interval_seconds=policy.retry_interval_in_seconds, + extensions=extensions, + ) + + +def _parameters_to_specs(parameters: dict[str, AdfParameter] | None) -> dict[str, ParameterSpec]: + """Map ADF pipeline parameters to shared parameter specs.""" + if not parameters: + return {} + return { + name: ParameterSpec(type=parameter.type, default=parameter.default_value) + for name, parameter in parameters.items() + } + + +def _variables_to_specs(variables: dict[str, AdfVariable] | None) -> dict[str, ParameterSpec]: + """Map ADF pipeline variables to shared parameter specs (same shape).""" + if not variables: + return {} + return { + name: ParameterSpec(type=variable.type, default=variable.default_value) for name, variable in variables.items() + } diff --git a/src/flowx/sources/adf/loader.py b/src/flowx/sources/adf/loader.py index 5fcc5d6..f1dff24 100644 --- a/src/flowx/sources/adf/loader.py +++ b/src/flowx/sources/adf/loader.py @@ -8,7 +8,8 @@ import logging import re import shutil -from datetime import datetime, timezone +import sys +from collections.abc import Mapping from pathlib import Path from typing import Any @@ -29,6 +30,9 @@ InventoryItem, TranslationStrategy, ) +from flowx.models.discovery import SourceGraph +from flowx.models.ir import Lineage, MotifAnnotation +from flowx.models.motifs import DetectedMotif logger = logging.getLogger(__name__) @@ -789,50 +793,6 @@ def _classify_activities( _classify_activities(pipeline_name, switch_children, items) -# --------------------------------------------------------------------------- -# Serialisation helpers -# --------------------------------------------------------------------------- - - -def _inventory_to_dict(inventory: Inventory, source_dir: str) -> dict[str, Any]: - """Serialise an :class:`Inventory` to a JSON-friendly dictionary. - - Args: - inventory: The inventory to serialise. - source_dir: Original source directory path (for provenance). - - Returns: - Dictionary suitable for ``json.dumps``. - """ - pipeline_map: dict[str, list[dict[str, Any]]] = {} - for item in inventory.items: - entry: dict[str, Any] = { - "name": item.activity_name, - "type": item.activity_type, - "strategy": item.strategy.value, - } - if item.depends_on: - entry["depends_on"] = item.depends_on - pipeline_map.setdefault(item.pipeline_name, []).append(entry) - - total = inventory.deterministic_count + inventory.agentic_count + inventory.unsupported_count - coverage_pct = round((inventory.deterministic_count + inventory.agentic_count) / total * 100, 1) if total else 0.0 - - return { - "source_dir": source_dir, - "generated_at": datetime.now(timezone.utc).isoformat(), - "pipelines": [{"name": pname, "activities": acts} for pname, acts in pipeline_map.items()], - "summary": { - "pipeline_count": inventory.pipeline_count, - "activity_count": total, - "deterministic_count": inventory.deterministic_count, - "agentic_count": inventory.agentic_count, - "unsupported_count": inventory.unsupported_count, - "coverage_pct": coverage_pct, - }, - } - - # --------------------------------------------------------------------------- # Profile complexity report (CSV) # --------------------------------------------------------------------------- @@ -924,7 +884,59 @@ def _tshirt_size(score: int) -> str: return "XL" -def build_profile_rows(definitions: AdfDefinitions) -> list[dict[str, Any]]: +def detect_motifs_by_pipeline(definitions: AdfDefinitions) -> dict[str, list[DetectedMotif]]: + """Detect the motifs in every pipeline, keyed by pipeline name. + + Runs the profiler's motif detector (:func:`flowx.motifs.detector.detect_motifs`) + once per pipeline, so the discover phase can both surface the full detected + motifs additively in ``inventory.json`` and count them in the profile report + from a single detection pass. Detection is best-effort: a pipeline whose + detection raises is logged and recorded with an empty list, never allowed to + hard-fail discover. This only *detects* motifs -- it never collapses their + member activities, which stays a convert-phase decision. + """ + from flowx.motifs.detector import detect_motifs + + results: dict[str, list[DetectedMotif]] = {} + for pipeline in definitions.pipelines: + try: + results[pipeline.name] = detect_motifs(pipeline, definitions) + except Exception as exc: # noqa: BLE001 - detection must never hard-fail discover + logger.warning("Motif detection failed for pipeline %r: %s", pipeline.name, exc) + results[pipeline.name] = [] + return results + + +def attach_motifs_to_graphs(graphs: list[SourceGraph], motifs_by_pipeline: Mapping[str, list[DetectedMotif]]) -> None: + """Record each pipeline's detected motifs on its source graph as lineage motif annotations. + + Motifs are deterministic facts about the pipeline, so they belong in the saved + graph next to its edges. Detection itself still runs on the ADF definitions; + this only copies the results onto the graph, member activities untouched. + """ + for graph in graphs: + detections = motifs_by_pipeline.get(graph.name) + if not detections: + continue + if graph.lineage is None: + graph.lineage = Lineage() + graph.lineage.motifs = [ + MotifAnnotation( + motif_id=motif.definition.motif_id, + member_task_keys=list(motif.matched_activities), + display_name=motif.definition.display_name, + databricks_replacement=motif.definition.databricks_replacement, + notes=list(motif.confidence_notes), + source_type_hint=motif.source_type_hint, + ) + for motif in detections + ] + + +def build_profile_rows( + definitions: AdfDefinitions, + motifs_by_pipeline: dict[str, list[DetectedMotif]] | None = None, +) -> list[dict[str, Any]]: """Builds one profile-report row per pipeline. Each row carries the source activity / dataset / linked-service counts, the @@ -933,20 +945,22 @@ def build_profile_rows(definitions: AdfDefinitions) -> list[dict[str, Any]]: Args: definitions: Parsed ADF definitions. + motifs_by_pipeline: Optional precomputed motif detections keyed by + pipeline name (see :func:`detect_motifs_by_pipeline`). Passing the + same map the inventory emitter uses keeps the report's pattern count + consistent with the surfaced motifs and avoids detecting twice; when + omitted, detection runs here. Returns: List of row dicts ordered by pipeline name. """ - from flowx.motifs.detector import detect_motifs + if motifs_by_pipeline is None: + motifs_by_pipeline = detect_motifs_by_pipeline(definitions) rows: list[dict[str, Any]] = [] for pipeline in sorted(definitions.pipelines, key=lambda p: p.name): activity_count, datasets, linked_services, category_counts = _pipeline_reference_counts(pipeline, definitions) - try: - n_patterns = len(detect_motifs(pipeline, definitions)) - except Exception as exc: # noqa: BLE001 - profiling must never hard-fail on motif detection - logger.warning("Motif detection failed for pipeline %r: %s", pipeline.name, exc) - n_patterns = 0 + n_patterns = len(motifs_by_pipeline.get(pipeline.name, [])) score = _complexity_score(category_counts, len(datasets), len(linked_services), n_patterns) rows.append( { @@ -1043,6 +1057,40 @@ def clear_stale_outputs(output_dir: Path) -> None: (output_dir / filename).unlink(missing_ok=True) +def _warn_no_pipelines_found(source_dir: Path) -> None: + """Prints a loud stderr warning when discover parses zero pipelines. + + The usual cause is pointing ``--adf-source-path`` at a directory that holds ARM-template + ``.json`` file(s) instead of the expected ADF export layout (a ``pipelines/`` folder), or + passing a directory when a single ARM template file was meant. We tailor the guidance to + whichever case we can detect, so an empty run never looks like a successful one. The message + names ``--adf-source-path`` -- the user-facing flag the discover runner takes -- not the + loader's internal ``--source-dir`` it normalises to. + """ + source_path = Path(source_dir) + top_level_json = sorted(source_path.glob("*.json")) if source_path.is_dir() else [] + + banner = "!" * 72 + lines = [banner, "WARNING: discover loaded 0 pipelines -- nothing was migrated."] + if top_level_json: + lines.append( + f"Found {len(top_level_json)} top-level .json file(s) in {source_path} but no " + "recognized ADF export layout (no 'pipelines/' directory)." + ) + lines.append( + "If these are ARM templates, pass the ARM template file directly as the " + "--adf-source-path (a single .json file), not the containing directory." + ) + else: + lines.append( + f"No ADF pipelines were found under {source_path}. Pass an ARM template file as " + "the --adf-source-path, or point --adf-source-path at a directory with the expected " + "ADF export layout ('pipelines/', 'datasets/', 'linked_services/', ...)." + ) + lines.append(banner) + print("\n".join(lines), file=sys.stderr) + + def main(argv: list[str] | None = None) -> int: """Discover-phase entry point: load ADF, build the inventory + profile report. @@ -1059,7 +1107,7 @@ def main(argv: list[str] | None = None) -> int: default=Path("./flowx_output"), help=( "Migration output directory. Profile artifacts are written into its " - "metadata/ subfolder (inventory.json, profile_report.csv, .arm.json)." + "metadata/ subfolder (inventory.json, source_graphs.json, profile_report.csv, .arm.json)." ), ) parser.add_argument( @@ -1075,6 +1123,13 @@ def main(argv: list[str] | None = None) -> int: definitions = load_adf_definitions(args.source_dir) logger.info("Loaded %d pipeline(s) from %s", len(definitions.pipelines), args.source_dir) + # A directory of ARM templates with no recognized ``pipelines/`` layout parses to zero + # pipelines and would otherwise finish with a success-shaped "Loaded 0 pipeline(s)". Fail + # loud (stderr) so the operator notices the export layout / path is wrong rather than + # trusting an empty-but-green run. + if not definitions.pipelines: + _warn_no_pipelines_found(args.source_dir) + # Filter to a single pipeline when --pipeline is specified if args.pipeline: matched = [pipeline for pipeline in definitions.pipelines if pipeline.name == args.pipeline] @@ -1095,7 +1150,13 @@ def main(argv: list[str] | None = None) -> int: ) logger.info("Filtered to pipeline: %s", args.pipeline) - inventory = build_inventory(definitions) + # Inventory JSON is projected from the discovery graph contract via the + # source-agnostic emitter (so ADF and Airflow emit one shape); imported here + # to avoid a module-level cycle (discovery_mapping imports this loader). + from flowx.discovery_inventory import build_source_inventory + from flowx.discovery_serde import SOURCE_GRAPHS_FILENAME, write_source_graphs + from flowx.models.discovery import SOURCE_ADF + from flowx.sources.adf.discovery_mapping import adf_definitions_to_source_graphs output_dir: Path = args.output_dir.resolve() clear_stale_outputs(output_dir) @@ -1103,11 +1164,31 @@ def main(argv: list[str] | None = None) -> int: metadata_dir.mkdir(parents=True, exist_ok=True) inventory_path = metadata_dir / "inventory.json" - inventory_dict = _inventory_to_dict(inventory, str(args.source_dir)) + source_graphs = adf_definitions_to_source_graphs(definitions) + # Detect motifs once and share the result: the graphs carry them (so they are saved and + # hashed with each graph), the inventory surfaces them, the profile report counts them. + motifs_by_pipeline = detect_motifs_by_pipeline(definitions) + attach_motifs_to_graphs(source_graphs, motifs_by_pipeline) + + # Persist the full graphs first, so the inventory projected from them can record which saved + # graph it describes, and a later phase can read exactly what discover saw. + source_graphs_path = metadata_dir / SOURCE_GRAPHS_FILENAME + source_graphs_document = write_source_graphs(source_graphs_path, source_graphs, source=SOURCE_ADF) + logger.info("Wrote source graphs to %s", source_graphs_path) + + inventory_dict = build_source_inventory( + source_graphs, + source=SOURCE_ADF, + source_dir=str(args.source_dir), + # ADF has historically omitted zero-activity pipelines from the per-pipeline + # listing while still counting them in summary.pipeline_count; preserve that. + include_empty_pipelines=False, + source_graphs_sha256=source_graphs_document["document_sha256"], + ) inventory_path.write_text(json.dumps(inventory_dict, indent=2), encoding="utf-8") logger.info("Wrote inventory to %s", inventory_path) - profile_rows = build_profile_rows(definitions) + profile_rows = build_profile_rows(definitions, motifs_by_pipeline) csv_path = metadata_dir / "profile_report.csv" write_profile_csv(profile_rows, csv_path) logger.info("Wrote profile report to %s", csv_path) @@ -1119,7 +1200,11 @@ def main(argv: list[str] | None = None) -> int: print("\nADF Profile Summary") print("===================") print(f"Pipelines parsed: {summary['pipeline_count']}") - print(f"Total activities: {summary['activity_count']}") + print(f"Total activities: {summary['activity_count']} (raw activities, incl. nested)") + # Discover counts every activity, including those nested inside ForEach/If/Switch. The + # convert phase reports top-level task-units *after* motif collapse, so its count is + # smaller -- label the units here so the two numbers are not mistaken for a discrepancy. + print(" (convert reports top-level task-units after motif collapse; expect fewer there)") print("\nStrategy Breakdown:") print(f" Deterministic: {summary['deterministic_count']}") print(f" Agentic: {summary['agentic_count']}") diff --git a/tests/resources/golden/adf_fixture_coverage.json b/tests/resources/golden/adf_fixture_coverage.json new file mode 100644 index 0000000..c36912d --- /dev/null +++ b/tests/resources/golden/adf_fixture_coverage.json @@ -0,0 +1,1162 @@ +[ + { + "activities": 13, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 13, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 31, + "complexity_size": "XL", + "control_flow_activities": 4, + "coverage_pct": 100.0, + "databricks_native_activities": 4, + "datasets": 3, + "deterministic_activities": 13, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 1, + "migration_status": "included", + "other_activities": 5, + "pipeline": "pipeline_all_activity_types", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 8, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 8, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 1, + "complexity_score": 23, + "complexity_size": "L", + "control_flow_activities": 3, + "coverage_pct": 100.0, + "databricks_native_activities": 1, + "datasets": 2, + "deterministic_activities": 8, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 1, + "migration_status": "included", + "other_activities": 4, + "pipeline": "pipeline_all_dependency_conditions", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 5, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 5, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 1, + "complexity_score": 14, + "complexity_size": "M", + "control_flow_activities": 3, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 1, + "deterministic_activities": 5, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 2, + "pipeline": "pipeline_append_variable_loop", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 8, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 8, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 3, + "complexity_score": 26, + "complexity_size": "L", + "control_flow_activities": 1, + "coverage_pct": 100.0, + "databricks_native_activities": 3, + "datasets": 4, + "deterministic_activities": 8, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 2, + "migration_status": "included", + "other_activities": 4, + "pipeline": "pipeline_complex_etl", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 15, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 15, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 1, + "complexity_score": 39, + "complexity_size": "XL", + "control_flow_activities": 7, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 0, + "deterministic_activities": 15, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 8, + "pipeline": "pipeline_complex_orchestration", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 1, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 1, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 7, + "complexity_size": "M", + "control_flow_activities": 0, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 2, + "deterministic_activities": 1, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 2, + "migration_status": "included", + "other_activities": 1, + "pipeline": "pipeline_copy_csv_to_delta", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 1, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 1, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 6, + "complexity_size": "M", + "control_flow_activities": 0, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 2, + "deterministic_activities": 1, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 1, + "migration_status": "included", + "other_activities": 1, + "pipeline": "pipeline_copy_parquet_to_delta", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 1, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 1, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 6, + "complexity_size": "M", + "control_flow_activities": 0, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 2, + "deterministic_activities": 1, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 1, + "migration_status": "included", + "other_activities": 1, + "pipeline": "pipeline_copy_sql_to_delta", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 2, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 2, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 8, + "complexity_size": "M", + "control_flow_activities": 0, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 2, + "deterministic_activities": 2, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 2, + "pipeline": "pipeline_delete_recursive", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 3, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 3, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 9, + "complexity_size": "M", + "control_flow_activities": 0, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 0, + "deterministic_activities": 3, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 3, + "pipeline": "pipeline_execute_pipeline_nested", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 4, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 4, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 10, + "complexity_size": "M", + "control_flow_activities": 2, + "coverage_pct": 100.0, + "databricks_native_activities": 1, + "datasets": 1, + "deterministic_activities": 4, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 1, + "migration_status": "included", + "other_activities": 1, + "pipeline": "pipeline_filter_array", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 7, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 7, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 18, + "complexity_size": "L", + "control_flow_activities": 3, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 0, + "deterministic_activities": 7, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 4, + "pipeline": "pipeline_foreach_switch", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 2, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 2, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 7, + "complexity_size": "M", + "control_flow_activities": 1, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 2, + "deterministic_activities": 2, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 1, + "pipeline": "pipeline_foreach_with_copy", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 5, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 5, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 15, + "complexity_size": "M", + "control_flow_activities": 1, + "coverage_pct": 100.0, + "databricks_native_activities": 1, + "datasets": 2, + "deterministic_activities": 5, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 1, + "migration_status": "included", + "other_activities": 3, + "pipeline": "pipeline_if_condition_branching", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 3, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 3, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 1, + "complexity_score": 12, + "complexity_size": "M", + "control_flow_activities": 1, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 3, + "deterministic_activities": 3, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 2, + "pipeline": "pipeline_lookup_and_foreach", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 6, + "agentic_activities": 4, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 6, + "code_attached_coverage_pct": 83.3, + "collapsible_patterns": 0, + "complexity_score": 19, + "complexity_size": "L", + "control_flow_activities": 0, + "coverage_pct": 83.3, + "databricks_native_activities": 0, + "datasets": 1, + "deterministic_activities": 1, + "deterministic_coverage_pct": 16.7, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 6, + "pipeline": "pipeline_mixed_agentic", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 4, + "unresolved_agentic_count": 0, + "unsupported_activities": 1 + }, + { + "activities": 6, + "agentic_activities": 1, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 6, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 1, + "complexity_score": 20, + "complexity_size": "L", + "control_flow_activities": 1, + "coverage_pct": 100.0, + "databricks_native_activities": 1, + "datasets": 3, + "deterministic_activities": 5, + "deterministic_coverage_pct": 83.3, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 1, + "migration_status": "included", + "other_activities": 4, + "pipeline": "pipeline_mixed_deterministic_agentic", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 1, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 1, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 1, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 2, + "complexity_size": "S", + "control_flow_activities": 0, + "coverage_pct": 100.0, + "databricks_native_activities": 1, + "datasets": 0, + "deterministic_activities": 1, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 1, + "migration_status": "included", + "other_activities": 0, + "pipeline": "pipeline_notebook_basic", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 1, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 1, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 2, + "complexity_size": "S", + "control_flow_activities": 0, + "coverage_pct": 100.0, + "databricks_native_activities": 1, + "datasets": 0, + "deterministic_activities": 1, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 1, + "migration_status": "included", + "other_activities": 0, + "pipeline": "pipeline_notebook_with_params", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 5, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 5, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 10, + "complexity_size": "M", + "control_flow_activities": 4, + "coverage_pct": 100.0, + "databricks_native_activities": 1, + "datasets": 0, + "deterministic_activities": 5, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 1, + "migration_status": "included", + "other_activities": 0, + "pipeline": "pipeline_set_variable_chain", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 1, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 1, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 2, + "complexity_size": "S", + "control_flow_activities": 0, + "coverage_pct": 100.0, + "databricks_native_activities": 1, + "datasets": 0, + "deterministic_activities": 1, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 1, + "migration_status": "included", + "other_activities": 0, + "pipeline": "pipeline_spark_jar_job", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 2, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 2, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 4, + "complexity_size": "S", + "control_flow_activities": 0, + "coverage_pct": 100.0, + "databricks_native_activities": 2, + "datasets": 0, + "deterministic_activities": 2, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 2, + "migration_status": "included", + "other_activities": 0, + "pipeline": "pipeline_spark_python_job", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 7, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 7, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 23, + "complexity_size": "L", + "control_flow_activities": 2, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 4, + "deterministic_activities": 7, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 5, + "pipeline": "pipeline_switch_multi_case", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 5, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 5, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 12, + "complexity_size": "M", + "control_flow_activities": 2, + "coverage_pct": 100.0, + "databricks_native_activities": 1, + "datasets": 0, + "deterministic_activities": 5, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 1, + "migration_status": "included", + "other_activities": 2, + "pipeline": "pipeline_wait_between_steps", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 3, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 3, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 9, + "complexity_size": "M", + "control_flow_activities": 0, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 0, + "deterministic_activities": 3, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 3, + "pipeline": "pipeline_web_activity_auth", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 3, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 3, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 6, + "complexity_size": "M", + "control_flow_activities": 3, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 0, + "deterministic_activities": 3, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 0, + "pipeline": "pl_test_appendvariable_coverage", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 2, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 2, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 10, + "complexity_size": "M", + "control_flow_activities": 0, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 4, + "deterministic_activities": 2, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 2, + "pipeline": "pl_test_copy_coverage", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 1, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 1, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 4, + "complexity_size": "S", + "control_flow_activities": 0, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 1, + "deterministic_activities": 1, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 1, + "pipeline": "pl_test_delete_coverage", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 2, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 2, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 6, + "complexity_size": "M", + "control_flow_activities": 0, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 0, + "deterministic_activities": 2, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 2, + "pipeline": "pl_test_executepipeline_coverage", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 4, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 4, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 10, + "complexity_size": "M", + "control_flow_activities": 3, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 1, + "deterministic_activities": 4, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 1, + "pipeline": "pl_test_filter_coverage", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 2, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 2, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 7, + "complexity_size": "M", + "control_flow_activities": 1, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 2, + "deterministic_activities": 2, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 1, + "pipeline": "pl_test_foreach_coverage", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 8, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 8, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 21, + "complexity_size": "L", + "control_flow_activities": 2, + "coverage_pct": 100.0, + "databricks_native_activities": 2, + "datasets": 2, + "deterministic_activities": 8, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 1, + "migration_status": "included", + "other_activities": 4, + "pipeline": "pl_test_ifcondition_coverage", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 2, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 2, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 7, + "complexity_size": "M", + "control_flow_activities": 0, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 1, + "deterministic_activities": 2, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 2, + "pipeline": "pl_test_lookup_coverage", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 1, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 1, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 2, + "complexity_size": "S", + "control_flow_activities": 0, + "coverage_pct": 100.0, + "databricks_native_activities": 1, + "datasets": 0, + "deterministic_activities": 1, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 1, + "migration_status": "included", + "other_activities": 0, + "pipeline": "pl_test_notebook_coverage", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 5, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 5, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 10, + "complexity_size": "M", + "control_flow_activities": 5, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 0, + "deterministic_activities": 5, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 0, + "pipeline": "pl_test_setvariable_coverage", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 1, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 1, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 2, + "complexity_size": "S", + "control_flow_activities": 0, + "coverage_pct": 100.0, + "databricks_native_activities": 1, + "datasets": 0, + "deterministic_activities": 1, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 1, + "migration_status": "included", + "other_activities": 0, + "pipeline": "pl_test_sparkjar_coverage", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 1, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 1, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 2, + "complexity_size": "S", + "control_flow_activities": 0, + "coverage_pct": 100.0, + "databricks_native_activities": 1, + "datasets": 0, + "deterministic_activities": 1, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 1, + "migration_status": "included", + "other_activities": 0, + "pipeline": "pl_test_sparkpython_coverage", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 6, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 6, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 10, + "complexity_size": "M", + "control_flow_activities": 1, + "coverage_pct": 100.0, + "databricks_native_activities": 4, + "datasets": 0, + "deterministic_activities": 6, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 1, + "migration_status": "included", + "other_activities": 1, + "pipeline": "pl_test_switch_coverage", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 2, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 2, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 4, + "complexity_size": "S", + "control_flow_activities": 2, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 0, + "deterministic_activities": 2, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 0, + "pipeline": "pl_test_wait_coverage", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + }, + { + "activities": 3, + "agentic_activities": 0, + "agentic_provider_version": "", + "agentic_resolution_outcomes": "{}", + "audited_activities": 3, + "code_attached_coverage_pct": 100.0, + "collapsible_patterns": 0, + "complexity_score": 9, + "complexity_size": "M", + "control_flow_activities": 0, + "coverage_pct": 100.0, + "databricks_native_activities": 0, + "datasets": 0, + "deterministic_activities": 3, + "deterministic_coverage_pct": 100.0, + "excluded_activities": 0, + "failed_activities": 0, + "finding_count": 0, + "finding_fingerprints": "[]", + "linked_services": 0, + "migration_status": "included", + "other_activities": 3, + "pipeline": "pl_test_webactivity_coverage", + "reconciliation_status": "not_applicable", + "resolved_agentic_count": 0, + "unresolved_agentic_count": 0, + "unsupported_activities": 0 + } +] \ No newline at end of file diff --git a/tests/unit/test_adf_dataset_lineage.py b/tests/unit/test_adf_dataset_lineage.py new file mode 100644 index 0000000..cc841a6 --- /dev/null +++ b/tests/unit/test_adf_dataset_lineage.py @@ -0,0 +1,903 @@ +"""Tests for the ADF dataset-lineage resolver (:mod:`flowx.sources.adf.dataset_lineage`). + +Proves the two-tier identity / signature model re-homed from the closed #36 +attempt: a table or a concrete path resolves to a physical ``identity``; a +parameterised reference does not (it never guesses) and falls back to the +structural path signature or the dataset name; and every input / output slot is +captured, not just index 0. +""" + +from __future__ import annotations + +from pathlib import Path + +from flowx.lineage import data_edges_from_endpoints +from flowx.models.adf_ast import ( + AdfActivity, + AdfDataset, + AdfDatasetReference, + AdfDefinitions, + AdfLinkedService, +) +from flowx.sources.adf.dataset_lineage import activity_data_assets, resolve_dataset_identity +from flowx.sources.adf.loader import load_adf_definitions + +FIXTURES_DIR = Path(__file__).parent.parent / "resources" / "json" + + +def _definitions(**datasets: AdfDataset) -> AdfDefinitions: + return AdfDefinitions(pipelines=[], datasets=dict(datasets)) + + +def _table_dataset(name: str, *, schema: str, table: str, linked_service: str = "ls_sql") -> AdfDataset: + return AdfDataset( + name=name, + type="AzureSqlTable", + properties={ + "typeProperties": {"schema": schema, "table": table}, + "linkedServiceName": {"referenceName": linked_service}, + }, + ) + + +def _adls_dataset( + name: str, *, file_system: str, folder_path: str, linked_service: str, file_name: object = None +) -> AdfDataset: + location: dict = {"fileSystem": file_system, "folderPath": folder_path} + if file_name is not None: + location["fileName"] = file_name + return AdfDataset( + name=name, + type="DelimitedText", + properties={ + "typeProperties": {"location": location}, + "linkedServiceName": {"referenceName": linked_service}, + }, + ) + + +def _adls_linked_service(name: str, *, account: str) -> AdfLinkedService: + return AdfLinkedService( + name=name, + type="AzureBlobFS", + properties={"typeProperties": {"url": f"https://{account}.dfs.core.windows.net"}}, + ) + + +def _sql_linked_service(name: str, *, connection_string: object) -> AdfLinkedService: + return AdfLinkedService( + name=name, + type="AzureSqlDatabase", + properties={"typeProperties": {"connectionString": connection_string}}, + ) + + +def _copy(name: str, *, source: str, sink: str) -> AdfActivity: + return AdfActivity( + name=name, + type="Copy", + type_properties={"source": {"referenceName": source}, "sink": {"referenceName": sink}}, + ) + + +# --------------------------------------------------------------------------- # +# Identity tier +# --------------------------------------------------------------------------- # + + +def test_identity_resolves_schema_and_table() -> None: + definitions = _definitions(ds_orders=_table_dataset("ds_orders", schema="curated", table="orders")) + identity = resolve_dataset_identity(AdfDatasetReference(reference_name="ds_orders"), definitions) + assert identity == "ls_sql/curated.orders" + + +def test_identity_resolves_storage_path_from_linked_service() -> None: + definitions = AdfDefinitions( + pipelines=[], + datasets={ + "ds_raw": _adls_dataset("ds_raw", file_system="data", folder_path="raw/customers", linked_service="ls") + }, + linked_services={"ls": _adls_linked_service("ls", account="contosolake")}, + ) + identity = resolve_dataset_identity(AdfDatasetReference(reference_name="ds_raw"), definitions) + assert identity == "abfss://data@contosolake.dfs.core.windows.net/raw/customers" + + +def test_files_in_the_same_folder_resolve_to_distinct_identities() -> None: + """``raw/orders.csv`` and ``raw/customers.csv`` are different assets, so no identity edge joins them.""" + definitions = AdfDefinitions( + pipelines=[], + datasets={ + "ds_orders": _adls_dataset( + "ds_orders", file_system="data", folder_path="raw", linked_service="ls", file_name="orders.csv" + ), + "ds_customers": _adls_dataset( + "ds_customers", file_system="data", folder_path="raw", linked_service="ls", file_name="customers.csv" + ), + "ds_orders_again": _adls_dataset( + "ds_orders_again", file_system="data", folder_path="raw", linked_service="ls", file_name="orders.csv" + ), + }, + linked_services={"ls": _adls_linked_service("ls", account="acct")}, + ) + _, copy1_writes = activity_data_assets(_copy("Copy1", source="ds_orders_again", sink="ds_orders"), definitions) + copy2_reads, _ = activity_data_assets(_copy("Copy2", source="ds_customers", sink="ds_orders_again"), definitions) + + assert copy1_writes[0].identity == "abfss://data@acct.dfs.core.windows.net/raw/orders.csv" + assert copy2_reads[0].identity == "abfss://data@acct.dfs.core.windows.net/raw/customers.csv" + assert copy1_writes[0].signature != copy2_reads[0].signature + same_file_identity = resolve_dataset_identity(AdfDatasetReference(reference_name="ds_orders_again"), definitions) + assert same_file_identity == copy1_writes[0].identity + + +def test_parameterised_file_name_has_no_path_identity() -> None: + """A per-entity file name is only known at run time, so the folder alone is never used as its identity.""" + definitions = AdfDefinitions( + pipelines=[], + datasets={ + "ds_expression": _adls_dataset( + "ds_expression", + file_system="data", + folder_path="raw", + linked_service="ls", + file_name={"value": "@dataset().entity", "type": "Expression"}, + ), + "ds_bare_expression": _adls_dataset( + "ds_bare_expression", + file_system="data", + folder_path="raw", + linked_service="ls", + file_name="@dataset().entity", + ), + }, + linked_services={"ls": _adls_linked_service("ls", account="acct")}, + ) + assert resolve_dataset_identity(AdfDatasetReference(reference_name="ds_expression"), definitions) is None + assert resolve_dataset_identity(AdfDatasetReference(reference_name="ds_bare_expression"), definitions) is None + + +def test_same_table_name_on_different_servers_resolves_to_distinct_identities() -> None: + """A source ``dbo.Orders`` and a warehouse ``dbo.Orders`` are different tables, so no identity edge joins them.""" + definitions = AdfDefinitions( + pipelines=[], + datasets={ + "ds_source_orders": _table_dataset( + "ds_source_orders", schema="dbo", table="Orders", linked_service="ls_src" + ), + "ds_warehouse_orders": _table_dataset( + "ds_warehouse_orders", schema="dbo", table="Orders", linked_service="ls_dw" + ), + "ds_lake": _adls_dataset("ds_lake", file_system="lake", folder_path="orders", linked_service="ls_lake"), + }, + linked_services={ + "ls_src": _sql_linked_service( + "ls_src", connection_string="Server=tcp:onprem-sql.contoso.local,1433;Database=Sales;User ID=etl" + ), + "ls_dw": _sql_linked_service( + "ls_dw", connection_string={"type": "SecureString", "value": "Data Source=dw.contoso.net;"} + ), + "ls_lake": _adls_linked_service("ls_lake", account="lake"), + }, + ) + copy_a_reads, _ = activity_data_assets(_copy("Copy A", source="ds_source_orders", sink="ds_lake"), definitions) + _, copy_b_writes = activity_data_assets(_copy("Copy B", source="ds_lake", sink="ds_warehouse_orders"), definitions) + + assert copy_a_reads[0].identity == "tcp:onprem-sql.contoso.local,1433/Sales/dbo.Orders" + assert copy_b_writes[0].identity == "dw.contoso.net/dbo.Orders" + assert copy_a_reads[0].signature != copy_b_writes[0].signature + + +def test_table_identity_falls_back_to_linked_service_name_when_server_is_secret() -> None: + """A Key Vault connection string hides the server, so the linked service name qualifies the table.""" + key_vault_connection = { + "type": "AzureKeyVaultSecret", + "store": {"referenceName": "ls_key_vault", "type": "LinkedServiceReference"}, + "secretName": "sql-connection", + } + definitions = AdfDefinitions( + pipelines=[], + datasets={"ds_orders": _table_dataset("ds_orders", schema="dbo", table="Orders", linked_service="ls_vaulted")}, + linked_services={"ls_vaulted": _sql_linked_service("ls_vaulted", connection_string=key_vault_connection)}, + ) + identity = resolve_dataset_identity(AdfDatasetReference(reference_name="ds_orders"), definitions) + assert identity == "ls_vaulted/dbo.Orders" + + +def test_parameterised_storage_account_has_no_path_identity() -> None: + """A generic ADLS linked service names its account per binding, so the account is never part of an identity.""" + generic_lake = AdfLinkedService( + name="ls_generic_lake", + type="AzureBlobFS", + properties={ + "typeProperties": {"url": "https://@{linkedService().accountName}.dfs.core.windows.net"}, + "parameters": {"accountName": {"type": "String"}}, + }, + ) + definitions = AdfDefinitions( + pipelines=[], + datasets={ + "ds_orders": _adls_dataset( + "ds_orders", + file_system="data", + folder_path="raw", + linked_service="ls_generic_lake", + file_name="orders.csv", + ) + }, + linked_services={"ls_generic_lake": generic_lake}, + ) + assert resolve_dataset_identity(AdfDatasetReference(reference_name="ds_orders"), definitions) is None + + +def test_parameterised_database_has_no_table_identity() -> None: + """A literal server with a parameterised database does not say which database holds the table.""" + definitions = AdfDefinitions( + pipelines=[], + datasets={ + "ds_property": _table_dataset("ds_property", schema="dbo", table="Orders", linked_service="ls_property"), + "ds_connection": _table_dataset( + "ds_connection", schema="dbo", table="Orders", linked_service="ls_connection" + ), + }, + linked_services={ + "ls_property": AdfLinkedService( + name="ls_property", + type="SqlServer", + properties={"typeProperties": {"server": "sql.contoso.net", "database": "@{linkedService().dbName}"}}, + ), + "ls_connection": _sql_linked_service( + "ls_connection", connection_string="Server=sql.contoso.net;Initial Catalog=@{linkedService().dbName}" + ), + }, + ) + assert resolve_dataset_identity(AdfDatasetReference(reference_name="ds_property"), definitions) is None + assert resolve_dataset_identity(AdfDatasetReference(reference_name="ds_connection"), definitions) is None + + +def test_generic_linked_service_called_with_different_bindings_has_no_table_identity() -> None: + """One generic SQL linked service bound to a source and a warehouse server never yields one shared identity.""" + generic_sql = AdfLinkedService( + name="ls_generic_sql", + type="AzureSqlDatabase", + properties={ + "typeProperties": { + "connectionString": { + "type": "AzureKeyVaultSecret", + "store": {"referenceName": "ls_key_vault", "type": "LinkedServiceReference"}, + "secretName": "@{linkedService().secretName}", + } + }, + "parameters": {"secretName": {"type": "String"}}, + }, + ) + + def _bound_orders(name: str, secret_name: str) -> AdfDataset: + return AdfDataset( + name=name, + type="AzureSqlTable", + properties={ + "typeProperties": {"schema": "dbo", "table": "Orders"}, + "linkedServiceName": { + "referenceName": "ls_generic_sql", + "parameters": {"secretName": secret_name}, + }, + }, + ) + + definitions = AdfDefinitions( + pipelines=[], + datasets={ + "ds_source_orders": _bound_orders("ds_source_orders", "source-sql"), + "ds_warehouse_orders": _bound_orders("ds_warehouse_orders", "warehouse-sql"), + }, + linked_services={"ls_generic_sql": generic_sql}, + ) + copy_reads, copy_writes = activity_data_assets( + _copy("Copy", source="ds_source_orders", sink="ds_warehouse_orders"), definitions + ) + + assert copy_reads[0].identity is None + assert copy_writes[0].identity is None + + +def test_table_without_linked_service_has_no_identity() -> None: + """With no backing store named, a bare ``schema.table`` is not provably one physical table.""" + dataset = AdfDataset( + name="ds_orphan", + type="AzureSqlTable", + properties={"typeProperties": {"schema": "dbo", "table": "Orders"}}, + ) + definitions = _definitions(ds_orphan=dataset) + assert resolve_dataset_identity(AdfDatasetReference(reference_name="ds_orphan"), definitions) is None + + +def test_identity_resolves_dataset_param_from_call_site_literal() -> None: + """A ``@dataset().table`` expression resolves when the call site passes a literal.""" + dataset = AdfDataset( + name="ds_param", + type="AzureSqlTable", + properties={ + "typeProperties": {"schema": "dbo", "table": "@dataset().tbl"}, + "parameters": {"tbl": {"type": "String"}}, + "linkedServiceName": {"referenceName": "ls_sql"}, + }, + ) + definitions = _definitions(ds_param=dataset) + reference = AdfDatasetReference(reference_name="ds_param", parameters={"tbl": "shipments"}) + assert resolve_dataset_identity(reference, definitions) == "ls_sql/dbo.shipments" + + +def test_parameterised_table_is_not_guessed() -> None: + """An unresolved ``@pipeline()`` expression yields no identity (never a guess, per #36).""" + dataset = AdfDataset( + name="ds_dyn", + type="AzureSqlTable", + properties={ + "typeProperties": {"schema": "dbo", "table": "@pipeline().parameters.tableName"}, + "linkedServiceName": {"referenceName": "ls_sql"}, + }, + ) + definitions = _definitions(ds_dyn=dataset) + assert resolve_dataset_identity(AdfDatasetReference(reference_name="ds_dyn"), definitions) is None + + +def test_unknown_dataset_reference_has_no_identity() -> None: + assert resolve_dataset_identity(AdfDatasetReference(reference_name="missing"), _definitions()) is None + + +# --------------------------------------------------------------------------- # +# activity_data_assets: reads/writes + the two-tier signature +# --------------------------------------------------------------------------- # + + +def test_copy_captures_source_read_and_sink_write_with_identity() -> None: + definitions = _definitions( + ds_src=_table_dataset("ds_src", schema="raw", table="orders"), + ds_dst=_table_dataset("ds_dst", schema="curated", table="orders"), + ) + activity = AdfActivity( + name="Copy Orders", + type="Copy", + type_properties={ + "source": {"referenceName": "ds_src"}, + "sink": {"referenceName": "ds_dst"}, + }, + ) + reads, writes = activity_data_assets(activity, definitions) + assert [(asset.identity, asset.signature) for asset in reads] == [("ls_sql/raw.orders", "ls_sql/raw.orders")] + assert [(asset.identity, asset.signature) for asset in writes] == [ + ("ls_sql/curated.orders", "ls_sql/curated.orders") + ] + assert reads[0].asset_type == "table" + + +def test_captures_all_inputs_and_outputs_not_just_index_zero() -> None: + """Every ``inputs`` / ``outputs`` slot is captured, not only the first.""" + definitions = _definitions( + ds_in_a=_table_dataset("ds_in_a", schema="raw", table="a"), + ds_in_b=_table_dataset("ds_in_b", schema="raw", table="b"), + ds_out_a=_table_dataset("ds_out_a", schema="curated", table="a"), + ds_out_b=_table_dataset("ds_out_b", schema="curated", table="b"), + ) + activity = AdfActivity( + name="Multi IO", + type="Copy", + inputs=[AdfDatasetReference(reference_name="ds_in_a"), AdfDatasetReference(reference_name="ds_in_b")], + outputs=[AdfDatasetReference(reference_name="ds_out_a"), AdfDatasetReference(reference_name="ds_out_b")], + ) + reads, writes = activity_data_assets(activity, definitions) + assert sorted(asset.identity for asset in reads) == ["ls_sql/raw.a", "ls_sql/raw.b"] + assert sorted(asset.identity for asset in writes) == ["ls_sql/curated.a", "ls_sql/curated.b"] + + +def test_dataset_named_in_both_slot_and_typeproperties_counted_once() -> None: + definitions = _definitions(ds_src=_table_dataset("ds_src", schema="raw", table="orders")) + activity = AdfActivity( + name="Lookup", + type="Lookup", + inputs=[AdfDatasetReference(reference_name="ds_src")], + type_properties={"dataset": {"referenceName": "ds_src"}}, + ) + reads, _ = activity_data_assets(activity, definitions) + assert [asset.identity for asset in reads] == ["ls_sql/raw.orders"] + + +def test_unresolved_reference_falls_back_to_path_signature() -> None: + """A parameterised file reference with a literal folder anchor gets a structural signature.""" + dataset = AdfDataset( + name="ds_wm", + type="DelimitedText", + properties={"typeProperties": {"location": {"fileSystem": "@dataset().fs"}}}, + ) + definitions = _definitions(ds_wm=dataset) + reference = AdfDatasetReference( + reference_name="ds_wm", + parameters={"folderPath": "@concat('watermark/', item().entity)", "fileName": "version.txt"}, + ) + activity = AdfActivity(name="Read WM", type="Lookup", type_properties={"dataset": _ref_dict(reference)}) + reads, _ = activity_data_assets(activity, definitions) + assert reads[0].identity is None + assert reads[0].signature == "FP[watermark|slots=1]/FN[version.txt|slots=0]" + + +def test_unresolved_reference_without_path_anchor_has_empty_signature() -> None: + """No identity and no path anchor -> empty signature, so it never joins (#36 name rule).""" + dataset = AdfDataset( + name="ds_generic", + type="AzureSqlTable", + properties={"typeProperties": {"table": "@pipeline().parameters.t"}}, + ) + definitions = _definitions(ds_generic=dataset) + activity = AdfActivity( + name="Copy", + type="Copy", + type_properties={"source": {"referenceName": "ds_generic"}}, + ) + reads, _ = activity_data_assets(activity, definitions) + assert reads[0].identity is None + # Never the bare dataset name: an empty signature cannot produce a signature-tier join. + assert reads[0].signature == "" + + +def test_distinct_parameter_bindings_of_same_dataset_are_all_retained() -> None: + """Same dataset ref, different params -> distinct physical assets, both kept (not collapsed).""" + dataset = AdfDataset( + name="ds", + type="AzureSqlTable", + properties={ + "typeProperties": {"schema": "raw", "table": "@dataset().tbl"}, + "parameters": {"tbl": {"type": "String"}}, + "linkedServiceName": {"referenceName": "ls_sql"}, + }, + ) + definitions = _definitions(ds=dataset) + activity = AdfActivity( + name="Multi Bind", + type="Copy", + inputs=[ + AdfDatasetReference(reference_name="ds", parameters={"tbl": "orders"}), + AdfDatasetReference(reference_name="ds", parameters={"tbl": "customers"}), + ], + ) + reads, _ = activity_data_assets(activity, definitions) + assert sorted(asset.identity for asset in reads) == ["ls_sql/raw.customers", "ls_sql/raw.orders"] + + +def test_same_ref_same_params_in_slot_and_typeproperties_still_collapses() -> None: + """The de-dup only fires on an identical binding: one asset, not two.""" + definitions = _definitions(ds_src=_table_dataset("ds_src", schema="raw", table="orders")) + activity = AdfActivity( + name="Lookup", + type="Lookup", + inputs=[AdfDatasetReference(reference_name="ds_src")], + type_properties={"dataset": {"referenceName": "ds_src"}}, + ) + reads, _ = activity_data_assets(activity, definitions) + assert [asset.identity for asset in reads] == ["ls_sql/raw.orders"] + + +def _ref_dict(reference: AdfDatasetReference) -> dict: + """Render a dataset reference the way ADF ``typeProperties`` nests it.""" + payload: dict = {"referenceName": reference.reference_name} + if reference.parameters: + payload["parameters"] = reference.parameters + return payload + + +def test_unevaluated_arm_expressions_never_become_identities() -> None: + """An ARM template expression the export left unevaluated is unresolved, so it never forms an identity.""" + definitions = AdfDefinitions( + pipelines=[], + datasets={ + "ds_table": _table_dataset("ds_table", schema="[parameters('schemaName')]", table="Orders"), + "ds_file": _adls_dataset( + "ds_file", file_system="[concat('raw', parameters('env'))]", folder_path="in", linked_service="ls" + ), + "ds_escaped": _adls_dataset("ds_escaped", file_system="[[literal]", folder_path="in", linked_service="ls"), + }, + linked_services={ + "ls_sql": AdfLinkedService( + name="ls_sql", type="AzureSqlDatabase", properties={"typeProperties": {"server": "sql.example.net"}} + ), + "ls": _adls_linked_service("ls", account="acct"), + }, + ) + assert resolve_dataset_identity(AdfDatasetReference(reference_name="ds_table"), definitions) is None + assert resolve_dataset_identity(AdfDatasetReference(reference_name="ds_file"), definitions) is None + # ARM's "[[" escape is a literal value that starts with "[", so it stays physical. + assert resolve_dataset_identity(AdfDatasetReference(reference_name="ds_escaped"), definitions) is not None + + +def test_bracket_quoted_sql_names_keep_their_identity() -> None: + """Bracket-quoted SQL names are literal table names, not ARM expressions, so they keep their identity.""" + definitions = AdfDefinitions( + pipelines=[], + datasets={ + "ds_legacy": AdfDataset( + name="ds_legacy", + type="AzureSqlTable", + properties={ + "typeProperties": {"tableName": "[dbo].[Orders]"}, + "linkedServiceName": {"referenceName": "ls_sql"}, + }, + ), + "ds_spaced": _table_dataset("ds_spaced", schema="dbo", table="[Order Details]"), + }, + linked_services={ + "ls_sql": AdfLinkedService( + name="ls_sql", type="AzureSqlDatabase", properties={"typeProperties": {"server": "sql.example.net"}} + ), + }, + ) + assert ( + resolve_dataset_identity(AdfDatasetReference(reference_name="ds_legacy"), definitions) + == "sql.example.net/[dbo].[Orders]" + ) + assert ( + resolve_dataset_identity(AdfDatasetReference(reference_name="ds_spaced"), definitions) + == "sql.example.net/dbo.[Order Details]" + ) + + +def test_arm_expression_storage_url_has_no_path_identity() -> None: + """An unevaluated ARM ``url`` is checked whole, so a cut-off fragment never stands in for the account.""" + arm_lake = AdfLinkedService( + name="ls_arm_lake", + type="AzureBlobFS", + properties={ + "typeProperties": {"url": "[concat('https://', parameters('storageAccountName'), '.dfs.core.windows.net')]"} + }, + ) + definitions = AdfDefinitions( + pipelines=[], + datasets={ + "ds_orders": _adls_dataset( + "ds_orders", file_system="raw", folder_path="in", linked_service="ls_arm_lake", file_name="orders.csv" + ) + }, + linked_services={"ls_arm_lake": arm_lake}, + ) + assert resolve_dataset_identity(AdfDatasetReference(reference_name="ds_orders"), definitions) is None + + +def _lake_definitions() -> AdfDefinitions: + return AdfDefinitions( + pipelines=[], + datasets={"ds_lake": _adls_dataset("ds_lake", file_system="data", folder_path="landing", linked_service="ls")}, + linked_services={"ls": _adls_linked_service("ls", account="acct")}, + ) + + +def test_store_settings_path_override_read_gets_no_identity_edge() -> None: + """A Copy that reads through a wildcard override does not read the dataset's folder, so no identity edge forms.""" + definitions = _lake_definitions() + ingest = AdfActivity( + name="Ingest", + type="Copy", + outputs=[AdfDatasetReference(reference_name="ds_lake")], + type_properties={"sink": {"type": "DelimitedTextSink"}}, + ) + publish = AdfActivity( + name="Publish", + type="Copy", + inputs=[AdfDatasetReference(reference_name="ds_lake")], + type_properties={ + "source": { + "type": "DelimitedTextSource", + "storeSettings": {"wildcardFolderPath": "archive/2023", "wildcardFileName": "*.csv"}, + } + }, + ) + _, ingest_writes = activity_data_assets(ingest, definitions) + publish_reads, _ = activity_data_assets(publish, definitions) + + assert ingest_writes[0].identity == "abfss://data@acct.dfs.core.windows.net/landing" + assert publish_reads[0].identity is None + edges = data_edges_from_endpoints( + [("Ingest", asset) for asset in ingest_writes], [("Publish", asset) for asset in publish_reads] + ) + assert edges == [] + + +def test_store_settings_path_override_covers_lookup_and_dataset_activities() -> None: + """Lookup overrides on its ``source``; Delete and GetMetadata override directly on ``typeProperties``.""" + definitions = _lake_definitions() + lookup = AdfActivity( + name="Lookup", + type="Lookup", + type_properties={ + "source": {"type": "DelimitedTextSource", "storeSettings": {"prefix": "orders_"}}, + "dataset": {"referenceName": "ds_lake"}, + }, + ) + get_metadata = AdfActivity( + name="GetMetadata", + type="GetMetadata", + type_properties={"dataset": {"referenceName": "ds_lake"}, "storeSettings": {"fileListPath": "lists/today.txt"}}, + ) + recursive_delete = AdfActivity( + name="Delete", + type="Delete", + type_properties={"dataset": {"referenceName": "ds_lake"}, "storeSettings": {"recursive": True}}, + ) + + assert [asset.identity for asset in activity_data_assets(lookup, definitions)[0]] == [None] + assert [asset.identity for asset in activity_data_assets(get_metadata, definitions)[0]] == [None] + assert [asset.identity for asset in activity_data_assets(recursive_delete, definitions)[0]] == [ + "abfss://data@acct.dfs.core.windows.net/landing" + ] + + +def test_source_query_read_gets_no_identity_edge() -> None: + """A Lookup that runs its own query reads what the query returns, not the dataset's table.""" + definitions = load_adf_definitions(FIXTURES_DIR) + load_orders = AdfActivity( + name="LoadOrders", + type="Copy", + outputs=[AdfDatasetReference(reference_name="ds_azure_sql_orders")], + type_properties={"sink": {"type": "AzureSqlSink"}}, + ) + get_watermark = AdfActivity( + name="GetWatermark", + type="Lookup", + type_properties={ + "source": {"type": "AzureSqlSource", "sqlReaderQuery": "SELECT MAX(ts) AS wm FROM dbo.watermarks"}, + "dataset": {"referenceName": "ds_azure_sql_orders"}, + }, + ) + _, load_orders_writes = activity_data_assets(load_orders, definitions) + get_watermark_reads, _ = activity_data_assets(get_watermark, definitions) + + assert load_orders_writes[0].identity == "ls_azure_sql/dbo.orders" + assert get_watermark_reads[0].identity is None + edges = data_edges_from_endpoints( + [("LoadOrders", asset) for asset in load_orders_writes], + [("GetWatermark", asset) for asset in get_watermark_reads], + ) + assert edges == [] + + +def test_stored_procedure_source_and_sink_get_no_identity() -> None: + """A stored procedure decides what is read or written, so neither side keeps the dataset's table identity.""" + definitions = _definitions(ds_orders=_table_dataset("ds_orders", schema="dbo", table="Orders")) + lookup = AdfActivity( + name="LookupStoredProc", + type="Lookup", + type_properties={ + "source": {"type": "SqlSource", "sqlReaderStoredProcedureName": "dbo.usp_get_orders"}, + "dataset": {"referenceName": "ds_orders"}, + }, + ) + copy = AdfActivity( + name="UpsertOrders", + type="Copy", + inputs=[AdfDatasetReference(reference_name="ds_orders")], + outputs=[AdfDatasetReference(reference_name="ds_orders")], + type_properties={ + "source": {"type": "SqlSource"}, + "sink": {"type": "SqlSink", "sqlWriterStoredProcedureName": "dbo.usp_upsert_orders"}, + }, + ) + copy_reads, copy_writes = activity_data_assets(copy, definitions) + + assert [asset.identity for asset in activity_data_assets(lookup, definitions)[0]] == [None] + assert [asset.identity for asset in copy_writes] == [None] + assert [asset.identity for asset in copy_reads] == ["ls_sql/dbo.Orders"] + + +def test_connector_specific_query_read_gets_no_identity_edge() -> None: + """An Oracle ``oracleReaderQuery`` overrides the read like ``sqlReaderQuery``; an empty one does not.""" + definitions = AdfDefinitions( + pipelines=[], + datasets={ + "ds_oracle_control": AdfDataset( + name="ds_oracle_control", + type="OracleTable", + properties={ + "typeProperties": {"schema": "HR", "table": "ETL_CONTROL"}, + "linkedServiceName": {"referenceName": "ls_oracle"}, + }, + ) + }, + linked_services={ + "ls_oracle": AdfLinkedService( + name="ls_oracle", + type="Oracle", + properties={"typeProperties": {"connectionString": "host=ora.example.net;port=1521;serviceName=HR"}}, + ) + }, + ) + stage_control = AdfActivity( + name="StageControl", + type="Copy", + outputs=[AdfDatasetReference(reference_name="ds_oracle_control")], + type_properties={"sink": {"type": "OracleSink"}}, + ) + + def _lookup(name: str, query: str) -> AdfActivity: + return AdfActivity( + name=name, + type="Lookup", + type_properties={ + "source": {"type": "OracleSource", "oracleReaderQuery": query}, + "dataset": {"referenceName": "ds_oracle_control"}, + }, + ) + + _, stage_control_writes = activity_data_assets(stage_control, definitions) + get_watermark_reads, _ = activity_data_assets( + _lookup("GetWatermark", "SELECT MAX(loaded_at) FROM HR.ETL_AUDIT"), definitions + ) + read_control_reads, _ = activity_data_assets(_lookup("ReadControl", ""), definitions) + + assert stage_control_writes[0].identity == "ls_oracle/HR.ETL_CONTROL" + assert get_watermark_reads[0].identity is None + assert ( + data_edges_from_endpoints( + [("StageControl", asset) for asset in stage_control_writes], + [("GetWatermark", asset) for asset in get_watermark_reads], + ) + == [] + ) + assert read_control_reads[0].identity == "ls_oracle/HR.ETL_CONTROL" + + +def test_parameterised_file_location_bound_to_literals_resolves_to_its_path() -> None: + """``@dataset().x`` location parts resolve against the call-site bindings, as table names do.""" + definitions = AdfDefinitions( + pipelines=[], + datasets={ + "ds_param": _adls_dataset( + "ds_param", + file_system="data", + folder_path={"value": "@dataset().folder", "type": "Expression"}, + linked_service="ls", + file_name="@dataset().entity", + ), + }, + linked_services={"ls": _adls_linked_service("ls", account="acct")}, + ) + orders = AdfDatasetReference(reference_name="ds_param", parameters={"folder": "raw", "entity": "orders.csv"}) + customers = AdfDatasetReference(reference_name="ds_param", parameters={"folder": "raw", "entity": "customers.csv"}) + unbound = AdfDatasetReference(reference_name="ds_param", parameters={"folder": "raw"}) + assert resolve_dataset_identity(orders, definitions) == "abfss://data@acct.dfs.core.windows.net/raw/orders.csv" + assert ( + resolve_dataset_identity(customers, definitions) == "abfss://data@acct.dfs.core.windows.net/raw/customers.csv" + ) + # An unbound file-name parameter is unknown, so the folder alone is never used as the identity. + assert resolve_dataset_identity(unbound, definitions) is None + + +def test_overridden_read_carries_no_signature_so_it_cannot_join_on_the_weak_tier() -> None: + """With the writer's identity unresolved, a shared dataset binding must not join a wildcard reader by signature.""" + definitions = AdfDefinitions( + pipelines=[], + datasets={ + "ds_param": _adls_dataset( + "ds_param", + file_system="data", + folder_path="@dataset().folderPath", + linked_service="ls_secret", + ) + }, + linked_services={ + "ls_secret": AdfLinkedService(name="ls_secret", type="AzureBlobFS", properties={"typeProperties": {}}) + }, + ) + binding = {"folderPath": {"value": "@concat('landing/', pipeline().parameters.run)", "type": "Expression"}} + ingest = AdfActivity( + name="Ingest", + type="Copy", + outputs=[AdfDatasetReference(reference_name="ds_param", parameters=binding)], + type_properties={"sink": {"type": "DelimitedTextSink"}}, + ) + publish = AdfActivity( + name="Publish", + type="Copy", + inputs=[AdfDatasetReference(reference_name="ds_param", parameters=binding)], + type_properties={"source": {"type": "DelimitedTextSource", "storeSettings": {"wildcardFileName": "*.csv"}}}, + ) + _, ingest_writes = activity_data_assets(ingest, definitions) + publish_reads, _ = activity_data_assets(publish, definitions) + + assert ingest_writes[0].identity is None + assert ingest_writes[0].signature # the writer still has a path-anchored signature to join on + assert publish_reads[0].identity is None + assert publish_reads[0].signature == "" + edges = data_edges_from_endpoints( + [("Ingest", asset) for asset in ingest_writes], [("Publish", asset) for asset in publish_reads] + ) + assert edges == [] + + +def _landing_copy(name: str, *, dataset: str, produced: bool) -> AdfActivity: + """A Copy that writes (or reads) *dataset* bound to a run-specific folder under ``landing/``.""" + reference = AdfDatasetReference( + reference_name=dataset, + parameters={"folderPath": {"value": "@concat('landing/', pipeline().parameters.run)", "type": "Expression"}}, + ) + if produced: + return AdfActivity(name=name, type="Copy", outputs=[reference]) + return AdfActivity(name=name, type="Copy", inputs=[reference]) + + +def _landing_definitions(**stores: tuple[str, str]) -> AdfDefinitions: + """One parameterised-folder dataset per entry, on container ``stores[name][0]`` of account ``stores[name][1]``.""" + return AdfDefinitions( + pipelines=[], + datasets={ + name: _adls_dataset( + name, file_system=container, folder_path="@dataset().folderPath", linked_service=f"ls_{account}" + ) + for name, (container, account) in stores.items() + }, + linked_services={ + f"ls_{account}": _adls_linked_service(f"ls_{account}", account=account) for _, account in stores.values() + }, + ) + + +def test_same_folder_shape_on_different_stores_does_not_join_by_signature() -> None: + """A write to sales@acctA and a read of hr@acctB share only a folder skeleton, so no edge joins them.""" + definitions = _landing_definitions(ds_sales=("sales", "acctA"), ds_hr=("hr", "acctB")) + _, sales_writes = activity_data_assets(_landing_copy("WriteSales", dataset="ds_sales", produced=True), definitions) + hr_reads, _ = activity_data_assets(_landing_copy("ReadHR", dataset="ds_hr", produced=False), definitions) + + assert sales_writes[0].identity is None + assert hr_reads[0].identity is None + edges = data_edges_from_endpoints( + [("WriteSales", asset) for asset in sales_writes], [("ReadHR", asset) for asset in hr_reads] + ) + assert edges == [] + + +def test_same_folder_shape_on_the_same_store_still_joins_by_signature() -> None: + """Two datasets on one literal container keep their advisory signature edge, now naming that store.""" + definitions = _landing_definitions(ds_out=("sales", "acctA"), ds_in=("sales", "acctA")) + _, writes = activity_data_assets(_landing_copy("Write", dataset="ds_out", produced=True), definitions) + reads, _ = activity_data_assets(_landing_copy("Read", dataset="ds_in", produced=False), definitions) + + edges = data_edges_from_endpoints([("Write", asset) for asset in writes], [("Read", asset) for asset in reads]) + assert [(edge.match_kind, edge.match_key) for edge in edges] == [ + ("signature", "ST[sales@acctA]/FP[landing|slots=1]/FN[None]") + ] + + +def _vaulted_blob_definitions(**containers: str) -> AdfDefinitions: + """One parameterised-folder dataset per entry, on container ``containers[name]`` of the Key Vault blob fixture.""" + vaulted_blob = load_adf_definitions(FIXTURES_DIR).get_linked_service("ls_azure_blob") + assert vaulted_blob is not None + return AdfDefinitions( + pipelines=[], + datasets={ + name: _adls_dataset( + name, file_system=container, folder_path="@dataset().folderPath", linked_service="ls_azure_blob" + ) + for name, container in containers.items() + }, + linked_services={"ls_azure_blob": vaulted_blob}, + ) + + +def test_same_folder_shape_on_different_containers_of_a_hidden_account_does_not_join() -> None: + """With the account in Key Vault, the linked service name still tells ``staging`` and ``archive`` apart.""" + definitions = _vaulted_blob_definitions(ds_staging="staging", ds_archive="archive") + _, staging_writes = activity_data_assets(_landing_copy("Stage", dataset="ds_staging", produced=True), definitions) + archive_reads, _ = activity_data_assets(_landing_copy("Archive", dataset="ds_archive", produced=False), definitions) + + edges = data_edges_from_endpoints( + [("Stage", asset) for asset in staging_writes], [("Archive", asset) for asset in archive_reads] + ) + assert edges == [] + + +def test_same_folder_shape_on_one_container_of_a_hidden_account_still_joins() -> None: + """Two datasets on the same container of a Key Vault blob linked service join on that container and service.""" + definitions = _vaulted_blob_definitions(ds_out="staging", ds_in="staging") + _, writes = activity_data_assets(_landing_copy("Write", dataset="ds_out", produced=True), definitions) + reads, _ = activity_data_assets(_landing_copy("Read", dataset="ds_in", produced=False), definitions) + + edges = data_edges_from_endpoints([("Write", asset) for asset in writes], [("Read", asset) for asset in reads]) + assert [(edge.match_kind, edge.match_key) for edge in edges] == [ + ("signature", "ST[staging@ls_azure_blob]/FP[landing|slots=1]/FN[None]") + ] diff --git a/tests/unit/test_adf_discovery_mapping.py b/tests/unit/test_adf_discovery_mapping.py new file mode 100644 index 0000000..f9efb59 --- /dev/null +++ b/tests/unit/test_adf_discovery_mapping.py @@ -0,0 +1,519 @@ +"""Tests for the ADF -> discovery graph mapper (:mod:`flowx.sources.adf.discovery_mapping`). + +Proves the mapping is 1:1 and lossless: every activity becomes one discovery +node, the ADF type is retained verbatim as ``native_type`` / ``original_type``, +the source dict is preserved on ``raw``, dependency edges keep all of their +outcome conditions, and control-flow nesting maps to labelled container branches. +""" + +from __future__ import annotations + +from flowx.discovery_inventory import STRATEGY_PROPERTY +from flowx.models.adf_ast import AdfDefinitions +from flowx.models.discovery import ( + SOURCE_ADF, + ContainerNode, + GapNode, + SourceGraph, + SourceNode, +) +from flowx.sources.adf.discovery_mapping import ( + adf_definitions_to_source_graphs, + adf_pipeline_to_source_graph, +) +from flowx.sources.adf.loader import _parse_pipeline_json, _parse_trigger_json + + +def _pipeline(activities: list[dict], **props) -> SourceGraph: + data = {"name": props.pop("name", "pl"), "properties": {"activities": activities, **props}} + return adf_pipeline_to_source_graph(_parse_pipeline_json(data)) + + +def test_activity_maps_1to1_retaining_type_and_raw() -> None: + """Each activity becomes one node with its ADF type and raw dict preserved.""" + raw_activity = { + "name": "Run Notebook", + "type": "DatabricksNotebook", + "typeProperties": {"notebookPath": "/nb"}, + } + graph = _pipeline([raw_activity]) + + assert len(graph.tasks) == 1 + node = graph.tasks[0] + assert isinstance(node, SourceNode) + assert node.source == SOURCE_ADF + assert node.name == "Run Notebook" + assert node.task_key == "Run Notebook" + # ADF type is retained verbatim -- native_type is the original_type source. + assert node.native_type == "DatabricksNotebook" + # Verbatim source dict preserved for lossless fallback. + assert node.raw == raw_activity + # Target-side strategy stashed in the properties seam (not a typed field). + assert node.properties[STRATEGY_PROPERTY] == "deterministic" + + +def test_dependencies_capture_all_conditions() -> None: + """A dependency edge keeps every outcome condition, not just the first.""" + activities = [ + {"name": "A", "type": "Wait", "typeProperties": {"waitTimeInSeconds": 1}}, + { + "name": "B", + "type": "DatabricksNotebook", + "dependsOn": [{"activity": "A", "dependencyConditions": ["Succeeded", "Skipped"]}], + }, + ] + graph = _pipeline(activities) + + node_b = next(node for node in graph.tasks if node.name == "B") + assert len(node_b.dependencies) == 1 + assert node_b.dependencies[0].upstream == "A" + assert node_b.dependencies[0].conditions == ["Succeeded", "Skipped"] + + +def test_no_motif_collapse_every_activity_is_a_node() -> None: + """Activities that a motif would merge each stay their own node (no collapse).""" + activities = [ + {"name": "Load", "type": "Copy"}, + {"name": "Notify", "type": "WebActivity", "dependsOn": [{"activity": "Load"}]}, + ] + graph = _pipeline(activities) + assert [node.name for node in graph.tasks] == ["Load", "Notify"] + + +def test_control_flow_maps_to_container_branches() -> None: + """An IfCondition maps to a ContainerNode with true/false branches nested.""" + activities = [ + { + "name": "Check", + "type": "IfCondition", + "typeProperties": { + "ifTrueActivities": [{"name": "T", "type": "DatabricksNotebook"}], + "ifFalseActivities": [{"name": "F", "type": "Wait"}], + }, + } + ] + graph = _pipeline(activities) + + container = graph.tasks[0] + assert isinstance(container, ContainerNode) + assert container.native_type == "IfCondition" + assert list(container.branches.keys()) == ["true", "false"] + assert container.branches["true"][0].name == "T" + assert container.branches["false"][0].name == "F" + + +def test_switch_maps_cases_and_default_to_branches() -> None: + """A Switch maps each case value plus default to its own labelled branch.""" + activities = [ + { + "name": "Route", + "type": "Switch", + "typeProperties": { + "cases": [ + {"value": "gold", "activities": [{"name": "G", "type": "Copy"}]}, + {"value": "silver", "activities": [{"name": "S", "type": "Copy"}]}, + ], + "defaultActivities": [{"name": "D", "type": "Wait"}], + }, + } + ] + graph = _pipeline(activities) + + container = graph.tasks[0] + assert isinstance(container, ContainerNode) + assert list(container.branches.keys()) == ["gold", "silver", "default"] + assert container.branches["gold"][0].name == "G" + assert container.branches["default"][0].name == "D" + + +def test_unsupported_activity_becomes_gap_node() -> None: + """An unsupported ADF type maps to a GapNode carrying the reason and raw.""" + activities = [{"name": "Weird", "type": "TotallyUnknownType"}] + graph = _pipeline(activities) + + node = graph.tasks[0] + assert isinstance(node, GapNode) + assert node.reason is not None and "TotallyUnknownType" in node.reason + assert node.raw == {"name": "Weird", "type": "TotallyUnknownType"} + assert node.properties[STRATEGY_PROPERTY] == "unsupported" + + +def test_policy_maps_retry_and_preserves_timeout_verbatim() -> None: + """Retry count/interval map to typed fields; the ISO timeout rides in extensions.""" + activities = [ + { + "name": "Copy", + "type": "Copy", + "policy": { + "timeout": "0.12:00:00", + "retry": 3, + "retryIntervalInSeconds": 30, + "secureInput": True, + }, + } + ] + graph = _pipeline(activities) + + policy = graph.tasks[0].policy + assert policy is not None + assert policy.max_retries == 3 + assert policy.retry_interval_seconds == 30 + # Timeout normalisation to seconds is a target concern -> preserved verbatim. + assert policy.extensions["timeout"] == "0.12:00:00" + assert policy.extensions["secure_input"] is True + + +def test_graph_carries_parameters_variables_tags_and_folder() -> None: + """Graph-level metadata maps onto the shared fields / properties seam.""" + data = { + "name": "pl", + "properties": { + "activities": [{"name": "N", "type": "DatabricksNotebook"}], + "parameters": {"env": {"type": "String", "defaultValue": "dev"}}, + "variables": {"counter": {"type": "Integer"}}, + "annotations": ["team-a", "prod"], + "folder": {"name": "ingest/bronze"}, + }, + } + graph = adf_pipeline_to_source_graph(_parse_pipeline_json(data)) + + assert graph.source == SOURCE_ADF + assert graph.parameters["env"].type == "String" + assert graph.parameters["env"].default == "dev" + assert graph.variables["counter"].type == "Integer" + assert graph.tags == ["team-a", "prod"] + assert graph.properties["folder"] == "ingest/bronze" + assert graph.raw is not None # verbatim pipeline dict preserved + + +def test_definitions_map_preserves_pipeline_order(adf_definitions) -> None: + """The definitions-level mapper yields one graph per pipeline, in order.""" + graphs = adf_definitions_to_source_graphs(adf_definitions) + assert [g.name for g in graphs] == [p.name for p in adf_definitions.pipelines] + assert all(g.source == SOURCE_ADF for g in graphs) + + +# --------------------------------------------------------------------------- +# Triggers / schedules (BLOCKING 1) +# --------------------------------------------------------------------------- + + +def _definitions_with_trigger(trigger: dict) -> AdfDefinitions: + pipeline = _parse_pipeline_json({"name": "pl_sched", "properties": {"activities": []}}) + return AdfDefinitions(pipelines=[pipeline], triggers=[_parse_trigger_json(trigger)]) + + +def test_schedule_trigger_lands_in_source_graph_schedule() -> None: + """A ScheduleTrigger referencing a pipeline populates that graph's schedule.""" + trigger = { + "name": "tr_daily", + "properties": { + "type": "ScheduleTrigger", + "typeProperties": { + "recurrence": {"frequency": "Day", "interval": 1, "timeZone": "UTC"}, + }, + "pipelines": [{"pipelineReference": {"referenceName": "pl_sched", "type": "PipelineReference"}}], + }, + } + graphs = adf_definitions_to_source_graphs(_definitions_with_trigger(trigger)) + + schedule = graphs[0].schedule + assert schedule is not None + assert schedule.kind == "schedule" + # Recurrence payload preserved verbatim as the expression. + assert schedule.expression == {"frequency": "Day", "interval": 1, "timeZone": "UTC"} + assert schedule.timezone == "UTC" + # Full trigger properties preserved losslessly in extensions. + assert schedule.extensions["trigger_name"] == "tr_daily" + assert schedule.extensions["trigger_type"] == "ScheduleTrigger" + assert "properties" in schedule.extensions + + +def test_tumbling_window_trigger_expression_from_type_properties() -> None: + """A TumblingWindowTrigger keeps its typeProperties as the expression.""" + trigger = { + "name": "tr_tumble", + "properties": { + "type": "TumblingWindowTrigger", + "typeProperties": {"frequency": "Hour", "interval": 1, "startTime": "2024-01-01T00:00:00Z"}, + "pipelines": [{"pipelineReference": {"referenceName": "pl_sched"}}], + }, + } + graphs = adf_definitions_to_source_graphs(_definitions_with_trigger(trigger)) + + schedule = graphs[0].schedule + assert schedule is not None + assert schedule.kind == "interval" + assert schedule.expression == {"frequency": "Hour", "interval": 1, "startTime": "2024-01-01T00:00:00Z"} + + +def test_unreferenced_pipeline_has_no_schedule() -> None: + """A pipeline no trigger references keeps ``schedule is None``.""" + trigger = { + "name": "tr_other", + "properties": { + "type": "ScheduleTrigger", + "typeProperties": {"recurrence": {"frequency": "Day", "interval": 1}}, + "pipelines": [{"pipelineReference": {"referenceName": "some_other_pipeline"}}], + }, + } + graphs = adf_definitions_to_source_graphs(_definitions_with_trigger(trigger)) + assert graphs[0].schedule is None + + +def test_multiple_triggers_preserve_extras_in_extensions() -> None: + """A second trigger for the same pipeline is preserved, not overwritten.""" + pipeline = _parse_pipeline_json({"name": "pl_sched", "properties": {"activities": []}}) + triggers = [ + _parse_trigger_json( + { + "name": "tr_first", + "properties": { + "type": "ScheduleTrigger", + "typeProperties": {"recurrence": {"frequency": "Day", "interval": 1}}, + "pipelines": [{"pipelineReference": {"referenceName": "pl_sched"}}], + }, + } + ), + _parse_trigger_json( + { + "name": "tr_second", + "properties": { + "type": "ScheduleTrigger", + "typeProperties": {"recurrence": {"frequency": "Hour", "interval": 6}}, + "pipelines": [{"pipelineReference": {"referenceName": "pl_sched"}}], + }, + } + ), + ] + graphs = adf_definitions_to_source_graphs(AdfDefinitions(pipelines=[pipeline], triggers=triggers)) + + schedule = graphs[0].schedule + assert schedule is not None + assert schedule.extensions["trigger_name"] == "tr_first" # first wins the typed slot + additional = schedule.extensions["additional_triggers"] + assert len(additional) == 1 + # The additional trigger retains its NAME (not just properties). + assert additional[0]["trigger_name"] == "tr_second" + assert additional[0]["trigger_type"] == "ScheduleTrigger" + assert additional[0]["properties"]["typeProperties"]["recurrence"]["interval"] == 6 + + +def test_triggers_do_not_leak_across_pipelines() -> None: + """A ScheduleSpec is per-pipeline: a later A-only trigger must not appear on B. + + Guards the shared-instance aliasing bug -- trigger_ab references A and B, then + trigger_a references only A. B must keep exactly trigger_ab and gain nothing + from trigger_a. + """ + pipeline_a = _parse_pipeline_json({"name": "pl_a", "properties": {"activities": []}}) + pipeline_b = _parse_pipeline_json({"name": "pl_b", "properties": {"activities": []}}) + triggers = [ + _parse_trigger_json( + { + "name": "tr_ab", + "properties": { + "type": "ScheduleTrigger", + "typeProperties": {"recurrence": {"frequency": "Day", "interval": 1}}, + "pipelines": [ + {"pipelineReference": {"referenceName": "pl_a"}}, + {"pipelineReference": {"referenceName": "pl_b"}}, + ], + }, + } + ), + _parse_trigger_json( + { + "name": "tr_a_only", + "properties": { + "type": "ScheduleTrigger", + "typeProperties": {"recurrence": {"frequency": "Hour", "interval": 2}}, + "pipelines": [{"pipelineReference": {"referenceName": "pl_a"}}], + }, + } + ), + ] + graphs = { + graph.name: graph + for graph in adf_definitions_to_source_graphs( + AdfDefinitions(pipelines=[pipeline_a, pipeline_b], triggers=triggers) + ) + } + + schedule_a = graphs["pl_a"].schedule + schedule_b = graphs["pl_b"].schedule + assert schedule_a is not None and schedule_b is not None + assert schedule_a is not schedule_b # distinct instances, no aliasing + # A picked up the second trigger; B must NOT have leaked it. + assert schedule_a.extensions["additional_triggers"][0]["trigger_name"] == "tr_a_only" + assert "additional_triggers" not in schedule_b.extensions + + +def test_fixture_scheduled_pipeline_gets_schedule(adf_definitions) -> None: + """End-to-end over the fixtures: a trigger-referenced pipeline gets a schedule.""" + graphs = {graph.name: graph for graph in adf_definitions_to_source_graphs(adf_definitions)} + # tr_daily_schedule references pipeline_copy_csv_to_delta in the fixtures. + assert graphs["pipeline_copy_csv_to_delta"].schedule is not None + + +# --------------------------------------------------------------------------- +# Empty / one-sided control flow (BLOCKING 2) +# --------------------------------------------------------------------------- + + +def test_empty_if_condition_stays_a_container_with_both_branches() -> None: + """An IfCondition with no children still maps to a ContainerNode, both branches present.""" + graph = _pipeline([{"name": "Gate", "type": "IfCondition", "typeProperties": {}}]) + + node = graph.tasks[0] + assert isinstance(node, ContainerNode) + assert node.native_type == "IfCondition" + assert list(node.branches.keys()) == ["true", "false"] + assert node.branches["true"] == [] + assert node.branches["false"] == [] + + +def test_empty_for_each_stays_a_container_with_body_branch() -> None: + """A ForEach with no children still maps to a ContainerNode with an empty body.""" + graph = _pipeline([{"name": "Loop", "type": "ForEach", "typeProperties": {}}]) + + node = graph.tasks[0] + assert isinstance(node, ContainerNode) + assert list(node.branches.keys()) == ["body"] + assert node.branches["body"] == [] + + +def test_empty_until_stays_a_container_with_body_branch() -> None: + """An Until with no children still maps to a ContainerNode with an empty body.""" + graph = _pipeline([{"name": "Retry", "type": "Until", "typeProperties": {}}]) + + node = graph.tasks[0] + assert isinstance(node, ContainerNode) + assert node.native_type == "Until" + assert list(node.branches.keys()) == ["body"] + assert node.branches["body"] == [] + + +def test_one_sided_if_keeps_empty_false_branch() -> None: + """An If with only a true branch keeps its false branch present-but-empty.""" + graph = _pipeline( + [ + { + "name": "Gate", + "type": "IfCondition", + "typeProperties": {"ifTrueActivities": [{"name": "T", "type": "Wait"}]}, + } + ] + ) + + node = graph.tasks[0] + assert isinstance(node, ContainerNode) + assert [child.name for child in node.branches["true"]] == ["T"] + assert "false" in node.branches # empty branch is present, not dropped + assert node.branches["false"] == [] + + +def test_switch_keeps_named_cases_that_have_no_activities() -> None: + """A declared case with ``activities: []`` keeps its label as an empty branch, in declaration order.""" + activities = [ + { + "name": "Route", + "type": "Switch", + "typeProperties": { + "cases": [ + {"value": "gold", "activities": [{"name": "G", "type": "Copy"}]}, + {"value": "skip", "activities": []}, + ], + "defaultActivities": [{"name": "D", "type": "Wait"}], + }, + } + ] + graph = _pipeline(activities) + + container = graph.tasks[0] + assert isinstance(container, ContainerNode) + assert list(container.branches.keys()) == ["gold", "skip", "default"] + assert container.branches["skip"] == [] + assert container.branches["gold"][0].name == "G" + + +def test_switch_case_without_a_value_keeps_its_declared_position() -> None: + """A case with no ``value`` is keyed like the parser keys it ("") and stays where it was declared.""" + activities = [ + { + "name": "Route", + "type": "Switch", + "typeProperties": { + "cases": [ + {"activities": [{"name": "A", "type": "Wait"}]}, + {"value": "x", "activities": []}, + ], + }, + } + ] + graph = _pipeline(activities) + + container = graph.tasks[0] + assert isinstance(container, ContainerNode) + assert list(container.branches.keys()) == ["", "x", "default"] + assert [child.name for child in container.branches[""]] == ["A"] + assert container.branches["x"] == [] + + +def test_switch_case_valued_default_does_not_collide_with_the_default_branch() -> None: + """A case whose value is literally "default" keeps its activities under ``case:default``.""" + activities = [ + { + "name": "Route", + "type": "Switch", + "typeProperties": { + "cases": [ + {"value": "default", "activities": [{"name": "CopyA", "type": "Copy"}]}, + {"value": "full", "activities": [{"name": "CopyB", "type": "Copy"}]}, + ], + "defaultActivities": [{"name": "Wait1", "type": "Wait"}], + }, + } + ] + graph = _pipeline(activities) + + container = graph.tasks[0] + assert isinstance(container, ContainerNode) + assert list(container.branches.keys()) == ["case:default", "full", "default"] + assert [child.name for child in container.branches["case:default"]] == ["CopyA"] + assert [child.name for child in container.branches["default"]] == ["Wait1"] + + +def test_switch_case_valued_default_does_not_collide_with_a_case_valued_case_default() -> None: + """A "default" case still keeps its activities when another case is literally valued "case:default".""" + activities = [ + { + "name": "Route", + "type": "Switch", + "typeProperties": { + "cases": [ + {"value": "default", "activities": [{"name": "CopyA", "type": "Copy"}]}, + {"value": "case:default", "activities": [{"name": "CopyC", "type": "Copy"}]}, + ], + "defaultActivities": [{"name": "Wait1", "type": "Wait"}], + }, + } + ] + graph = _pipeline(activities) + + container = graph.tasks[0] + assert isinstance(container, ContainerNode) + assert list(container.branches.keys()) == ["case:case:default", "case:default", "default"] + assert [child.name for child in container.branches["case:case:default"]] == ["CopyA"] + assert [child.name for child in container.branches["case:default"]] == ["CopyC"] + assert [child.name for child in container.branches["default"]] == ["Wait1"] + + +def test_empty_switch_stays_a_container_with_default_branch() -> None: + """A Switch with no cases still maps to a ContainerNode with an empty default.""" + graph = _pipeline([{"name": "Route", "type": "Switch", "typeProperties": {}}]) + + node = graph.tasks[0] + assert isinstance(node, ContainerNode) + assert list(node.branches.keys()) == ["default"] + assert node.branches["default"] == [] diff --git a/tests/unit/test_adf_inventory_superset.py b/tests/unit/test_adf_inventory_superset.py new file mode 100644 index 0000000..b82435c --- /dev/null +++ b/tests/unit/test_adf_inventory_superset.py @@ -0,0 +1,227 @@ +"""Consumer-safety tests for the ADF inventory emitted via the discovery graph + emitter. + +Two guarantees: + +* **Superset** -- the new ``inventory.json`` keeps every key the historical shape + had (per-activity ``name`` / ``type`` / ``strategy`` / ``depends_on`` and the + ``summary`` count block) byte-compatibly, and only *adds* fields on top. The + historical values are reconstructed here from :func:`build_inventory` (the + authoritative classifier, untouched by this slice). +* **Golden coverage** -- feeding the new inventory through the real consumer + (:func:`flowx.reporting.coverage.build_coverage_rows`) reproduces a committed + snapshot, so downstream reporting is provably unchanged. +""" + +from __future__ import annotations + +import json +from pathlib import Path + +from flowx.discovery_serde import read_source_graphs +from flowx.reporting.coverage import build_coverage_rows +from flowx.sources.adf.loader import build_inventory, load_adf_definitions, main + +FIXTURES_DIR = Path(__file__).parent.parent / "resources" / "json" +GOLDEN_COVERAGE = Path(__file__).parent.parent / "resources" / "golden" / "adf_fixture_coverage.json" + + +def _run_discover(tmp_path: Path) -> Path: + """Run the ADF discover entry point against the fixtures; return metadata dir.""" + exit_code = main(["--source-dir", str(FIXTURES_DIR), "--output-dir", str(tmp_path)]) + assert exit_code == 0 + return tmp_path / "metadata" + + +def _legacy_pipeline_map() -> dict[str, list[dict]]: + """Reconstruct the pre-change per-pipeline activity shape from build_inventory. + + This is exactly what the removed ``_inventory_to_dict`` used to emit, rebuilt + from the authoritative classifier so the superset check does not depend on a + frozen copy of the old serializer. + """ + inventory = build_inventory(load_adf_definitions(FIXTURES_DIR)) + pipeline_map: dict[str, list[dict]] = {} + for item in inventory.items: + entry: dict = {"name": item.activity_name, "type": item.activity_type, "strategy": item.strategy.value} + if item.depends_on: + entry["depends_on"] = item.depends_on + pipeline_map.setdefault(item.pipeline_name, []).append(entry) + return pipeline_map + + +def test_inventory_top_level_is_superset(tmp_path: Path) -> None: + """Top-level keys include the legacy set plus the new ``source`` discriminator.""" + metadata = _run_discover(tmp_path) + inventory = json.loads((metadata / "inventory.json").read_text()) + + # Legacy top-level keys still present. + for key in ("source_dir", "pipelines", "summary"): + assert key in inventory + # Additive discriminator. + assert inventory["source"] == "adf" + + +def test_summary_counts_match_legacy_classifier(tmp_path: Path) -> None: + """The summary count block is byte-compatible with the legacy classifier.""" + metadata = _run_discover(tmp_path) + summary = json.loads((metadata / "inventory.json").read_text())["summary"] + + legacy = build_inventory(load_adf_definitions(FIXTURES_DIR)) + total = legacy.deterministic_count + legacy.agentic_count + legacy.unsupported_count + assert summary["pipeline_count"] == legacy.pipeline_count + assert summary["activity_count"] == total + assert summary["deterministic_count"] == legacy.deterministic_count + assert summary["agentic_count"] == legacy.agentic_count + assert summary["unsupported_count"] == legacy.unsupported_count + assert summary["coverage_pct"] == round((legacy.deterministic_count + legacy.agentic_count) / total * 100, 1) + + +def test_activity_entries_superset_legacy_shape(tmp_path: Path) -> None: + """Every activity keeps its legacy keys/values and only gains additive fields.""" + metadata = _run_discover(tmp_path) + inventory = json.loads((metadata / "inventory.json").read_text()) + legacy_map = _legacy_pipeline_map() + + # Same pipeline membership (ADF omits zero-activity pipelines; fixtures have none empty). + new_names = [pipeline["name"] for pipeline in inventory["pipelines"]] + assert sorted(new_names) == sorted(legacy_map) + + for pipeline in inventory["pipelines"]: + legacy_entries = legacy_map[pipeline["name"]] + new_entries = pipeline["activities"] + assert len(new_entries) == len(legacy_entries) + for legacy_entry, new_entry in zip(legacy_entries, new_entries): + # Every legacy key/value survives byte-for-byte. + for key, value in legacy_entry.items(): + assert new_entry[key] == value, (pipeline["name"], key) + # Additive standardized fields are present. + assert new_entry["original_type"] == legacy_entry["type"] + assert "dependencies" in new_entry + assert "raw" in new_entry + + +def test_dependencies_field_carries_conditions(tmp_path: Path) -> None: + """The additive ``dependencies`` field carries upstream + conditions per edge.""" + metadata = _run_discover(tmp_path) + inventory = json.loads((metadata / "inventory.json").read_text()) + + # The all-dependency-conditions fixture exercises non-default conditions. + pipeline = next(p for p in inventory["pipelines"] if p["name"] == "pipeline_all_dependency_conditions") + edges = [dependency for activity in pipeline["activities"] for dependency in activity["dependencies"]] + assert edges, "expected at least one dependency edge" + for edge in edges: + assert set(edge.keys()) == {"upstream", "conditions", "resolved"} + assert isinstance(edge["conditions"], list) + + +def test_inventory_carries_per_pipeline_lineage(tmp_path: Path) -> None: + """The emitted inventory surfaces the lineage #62b derived over the ADF fixtures. + + Lineage rides additively on each pipeline entry, so the aggregate across the + per-pipeline blocks must match what the discovery lineage pass computed: on + these fixtures that is 11 cross-pipeline control edges and 0 data edges. + """ + metadata = _run_discover(tmp_path) + inventory = json.loads((metadata / "inventory.json").read_text()) + + control_edges = 0 + data_edges = 0 + for pipeline in inventory["pipelines"]: + assert "lineage" in pipeline, pipeline["name"] + # Additive block only -- historical per-pipeline keys are untouched. + assert set(pipeline["lineage"].keys()) == {"control_edges", "data_edges", "motifs"} + control_edges += len(pipeline["lineage"]["control_edges"]) + data_edges += len(pipeline["lineage"]["data_edges"]) + + assert control_edges == 11 + assert data_edges == 0 + + +def test_inventory_surfaces_detected_motifs_without_collapsing(tmp_path: Path) -> None: + """Discover surfaces the profiler's detected motifs additively, members intact. + + ``pipeline_complex_etl`` carries a detectable metadata-driven bulk-copy motif. + It must appear in the pipeline's additive ``motifs`` list -- carrying its type, + Databricks replacement target, and participating activities -- while every + member still appears as its own entry in ``activities`` (no discover-time + collapse) and the lineage block's own motif slot stays empty (decoupled). + """ + metadata = _run_discover(tmp_path) + inventory = json.loads((metadata / "inventory.json").read_text()) + + pipeline = next(p for p in inventory["pipelines"] if p["name"] == "pipeline_complex_etl") + bulk = next(m for m in pipeline["motifs"] if m["motif_id"] == "metadata_driven_bulk_copy") + + assert set(bulk.keys()) == { + "motif_id", + "display_name", + "databricks_replacement", + "member_task_keys", + "source_type_hint", + "confidence_notes", + } + assert bulk["databricks_replacement"] == "for_each_ingestion" + # Exact expected member set for this fixture -- a partial-member regression must fail. + assert bulk["member_task_keys"] == ["Lookup ETL Config", "ForEach Table In Config"] + + # No collapse at discover: each member survives as its own activity entry. + activity_names = {activity["name"] for activity in pipeline["activities"]} + for member in bulk["member_task_keys"]: + assert member in activity_names, member + + # Decoupled from lineage: the additive motifs key is separate from lineage.motifs. + assert pipeline["lineage"]["motifs"] == [] + + +def test_pipelines_without_a_motif_omit_the_motifs_key(tmp_path: Path) -> None: + """The motifs field is additive: a pipeline with no detected motif omits it.""" + metadata = _run_discover(tmp_path) + inventory = json.loads((metadata / "inventory.json").read_text()) + + plain = next(p for p in inventory["pipelines"] if p["name"] == "pipeline_notebook_basic") + assert "motifs" not in plain + + +def test_coverage_output_matches_golden(tmp_path: Path) -> None: + """The real coverage consumer reproduces the committed golden snapshot. + + This is the no-consumer-breakage guarantee: regenerate the golden with + ``make test`` after an intentional change and review the diff. + """ + metadata = _run_discover(tmp_path) + rows = json.loads(json.dumps(build_coverage_rows(metadata), sort_keys=True)) + golden = json.loads(GOLDEN_COVERAGE.read_text()) + assert rows == golden + + +def test_discover_persists_source_graphs_that_match_the_inventory(tmp_path: Path) -> None: + """Discover writes the full, hashed source graphs beside the inventory, one per parsed pipeline.""" + metadata = _run_discover(tmp_path) + + document = json.loads((metadata / "source_graphs.json").read_text(encoding="utf-8")) + graphs = read_source_graphs(metadata / "source_graphs.json") + inventory = json.loads((metadata / "inventory.json").read_text(encoding="utf-8")) + + assert document["contract_version"] == "1" + assert document["source"] == "adf" + assert len(document["graph_sha256"]) == len(graphs) == inventory["summary"]["pipeline_count"] + listed = {pipeline["name"] for pipeline in inventory["pipelines"]} + assert listed <= {graph.name for graph in graphs} + assert inventory["source_graphs_sha256"] == document["document_sha256"] + + +def test_detected_motifs_are_saved_inside_the_hashed_graphs(tmp_path: Path) -> None: + """Every motif the inventory lists is also recorded on that pipeline's saved graph.""" + metadata = _run_discover(tmp_path) + + graphs = {graph.name: graph for graph in read_source_graphs(metadata / "source_graphs.json")} + inventory = json.loads((metadata / "inventory.json").read_text(encoding="utf-8")) + + with_motifs = [pipeline for pipeline in inventory["pipelines"] if pipeline.get("motifs")] + assert with_motifs, "the ADF fixtures are expected to contain at least one detected motif" + for pipeline in with_motifs: + lineage = graphs[pipeline["name"]].lineage + assert lineage is not None + saved = sorted((motif.motif_id, tuple(motif.member_task_keys)) for motif in lineage.motifs) + listed = sorted((motif["motif_id"], tuple(motif["member_task_keys"])) for motif in pipeline["motifs"]) + assert set(listed) <= set(saved) diff --git a/tests/unit/test_adf_loader.py b/tests/unit/test_adf_loader.py index e673051..248f3e8 100644 --- a/tests/unit/test_adf_loader.py +++ b/tests/unit/test_adf_loader.py @@ -20,6 +20,7 @@ classify_activity, clear_stale_outputs, load_adf_definitions, + main, ) # --------------------------------------------------------------------------- @@ -554,3 +555,35 @@ def test_idempotent_on_empty_dir(self, tmp_path): """Clearing a directory with no flowx artifacts is a no-op (no error).""" clear_stale_outputs(tmp_path) assert list(tmp_path.iterdir()) == [] + + +class TestZeroPipelineWarning: + """discover must fail loud (stderr) when it parses zero pipelines.""" + + def test_arm_template_directory_warns_loudly(self, tmp_path, capsys): + """A dir holding ARM-template .json (no pipelines/ layout) loads 0 pipelines + warns.""" + source_dir = tmp_path / "arm_export" + source_dir.mkdir() + # An ARM template dropped at the top level, not the expected pipelines/ layout. + (source_dir / "ARMTemplateForFactory.json").write_text(json.dumps({"resources": []}), encoding="utf-8") + + exit_code = main(["--source-dir", str(source_dir), "--output-dir", str(tmp_path / "out")]) + + # Behaviour for the 0-pipeline case is unchanged (still exits 0); the deliverable is the + # loud stderr warning that steers the operator to pass the ARM template as the file path. + assert exit_code == 0 + stderr = capsys.readouterr().err + assert "0 pipelines" in stderr + assert "ARM template" in stderr + assert ".json" in stderr + # The guidance names the user-facing flag (--adf-source-path), not the loader-internal one. + assert "--adf-source-path" in stderr + assert "--source-dir" not in stderr + + def test_valid_directory_does_not_warn(self, fixtures_dir, tmp_path, capsys): + """A valid ADF export loads pipelines and emits no zero-pipeline warning.""" + exit_code = main(["--source-dir", str(fixtures_dir), "--output-dir", str(tmp_path / "out")]) + + assert exit_code == 0 + stderr = capsys.readouterr().err + assert "discover loaded 0 pipelines" not in stderr diff --git a/tests/unit/test_lineage_substrate.py b/tests/unit/test_lineage_substrate.py new file mode 100644 index 0000000..b32a737 --- /dev/null +++ b/tests/unit/test_lineage_substrate.py @@ -0,0 +1,487 @@ +"""Unit tests for the source-neutral lineage substrate (#61). + +Covers the pure derivation (:mod:`flowx.lineage`) -- control fan-out, nested and +Switch recursion, the identity-vs-signature join tiers, no self-edges, no +duplicate edges -- and the ``ir_serde`` round-trip for the new IR types plus the +new ``Activity`` base fields, including a MotifActivity round-trip that proves the +R1 ``motif_id`` collision is handled. +""" + +from __future__ import annotations + +import json + +from flowx.bundler.dab_writer import pipeline_dict_to_ir +from flowx.ir_serde import pipeline_to_dict +from flowx.lineage import ( + build_control_edges, + build_data_edges, + build_lineage, + build_motif_annotations, + with_lineage, +) +from flowx.models.ir import ( + ControlEdge, + DataAsset, + DataEdge, + ExecutePipelineActivity, + ForEachActivity, + IfConditionActivity, + Lineage, + MotifActivity, + MotifAnnotation, + NotebookActivity, + Pipeline, + RunJobActivity, + SwitchActivity, + SwitchCase, + WaitActivity, +) + + +def _notebook(task_key: str, *, reads=None, writes=None, motif_id=None) -> NotebookActivity: + return NotebookActivity( + name=task_key, + task_key=task_key, + notebook_path=f"/Shared/{task_key}", + data_reads=list(reads or []), + data_writes=list(writes or []), + motif_id=motif_id, + ) + + +def _execute(task_key: str, callee: str, *, wait: bool = True) -> ExecutePipelineActivity: + return ExecutePipelineActivity(name=task_key, task_key=task_key, pipeline_name=callee, wait_on_completion=wait) + + +# --------------------------------------------------------------------------- # +# Control-edge derivation +# --------------------------------------------------------------------------- # + + +def test_control_edges_fan_out_and_nested_switch_recursion(): + """ExecutePipeline calls are found at top level and inside ForEach/If/Switch.""" + pipeline = Pipeline( + name="parent", + tasks=[ + _execute("call_a", "child_a"), + ForEachActivity( + name="fe", + task_key="fe", + items_expression="@x", + inner_activities=[_execute("call_b", "child_b")], + ), + IfConditionActivity( + name="cond", + task_key="cond", + op="equals", + left="@a", + right="@b", + if_true_activities=[_execute("call_c", "child_c")], + if_false_activities=[_execute("call_d", "child_d")], + ), + SwitchActivity( + name="sw", + task_key="sw", + on_expression="@e", + cases=[SwitchCase(value="one", activities=[_execute("call_e", "child_e")])], + default_activities=[_execute("call_f", "child_f")], + ), + ], + ) + + edges = build_control_edges(pipeline) + + targets = sorted(edge.target_workflow for edge in edges) + assert targets == ["child_a", "child_b", "child_c", "child_d", "child_e", "child_f"] + assert all(edge.source_workflow == "parent" for edge in edges) + # Each call site keeps its own via_task_key (fan-out preserved). + assert {edge.via_task_key for edge in edges} == { + "call_a", + "call_b", + "call_c", + "call_d", + "call_e", + "call_f", + } + + +def test_control_edges_run_job_activity_is_source_neutral(): + """A RunJobActivity (Airflow) produces a control edge just like ExecutePipeline.""" + pipeline = Pipeline( + name="dag_main", + tasks=[RunJobActivity(name="run", task_key="run", job_name="downstream_job")], + ) + + edges = build_control_edges(pipeline) + + assert len(edges) == 1 + assert edges[0].source_workflow == "dag_main" + assert edges[0].target_workflow == "downstream_job" + assert edges[0].via_task_key == "run" + assert edges[0].wait_for_completion is None + assert edges[0].resolved is True + + +def test_run_now_of_an_existing_job_is_not_a_cross_workflow_edge(): + """A run-now by job ID targets a job outside the source; its job_name is only the caller's task key.""" + pipeline = Pipeline( + name="dag_main", + tasks=[RunJobActivity(name="trigger", task_key="trigger", job_name="trigger", existing_job_id="123")], + ) + + assert build_control_edges(pipeline) == [] + + +def test_control_edges_unresolved_callee_is_recorded_not_dropped(): + """An empty callee is kept with resolved=False rather than silently dropped.""" + pipeline = Pipeline(name="parent", tasks=[_execute("call", "")]) + + edges = build_control_edges(pipeline) + + assert len(edges) == 1 + assert edges[0].target_workflow == "" + assert edges[0].resolved is False + + +def test_control_edges_no_self_edge(): + """A pipeline invoking itself produces no edge.""" + pipeline = Pipeline(name="loop", tasks=[_execute("call", "loop")]) + + assert build_control_edges(pipeline) == [] + + +def test_control_edges_no_duplicate_from_recursion(): + """A single call site nested in a container is emitted exactly once.""" + pipeline = Pipeline( + name="parent", + tasks=[ + ForEachActivity( + name="fe", + task_key="fe", + items_expression="@x", + inner_activities=[_execute("call", "child")], + ) + ], + ) + + edges = build_control_edges(pipeline) + + assert len(edges) == 1 + assert edges[0].target_workflow == "child" + + +# --------------------------------------------------------------------------- # +# Data-edge derivation: identity vs signature tiers +# --------------------------------------------------------------------------- # + + +def test_data_edges_identity_tier_joins_across_different_signatures(): + """Two assets with the same resolved identity match even when their names differ.""" + pipeline = Pipeline( + name="p", + tasks=[ + _notebook("writer", writes=[DataAsset(signature="ds_out", identity="curated.orders")]), + _notebook("reader", reads=[DataAsset(signature="ds_in_other_name", identity="curated.orders")]), + ], + ) + + edges = build_data_edges(pipeline) + + assert len(edges) == 1 + assert edges[0].source_task_key == "writer" + assert edges[0].target_task_key == "reader" + assert edges[0].match_kind == "identity" + assert edges[0].match_key == "curated.orders" + assert edges[0].identity == "curated.orders" + + +def test_data_edges_signature_tier_when_identity_unresolved(): + """When identity is unresolvable, matching falls back to the neutral signature.""" + pipeline = Pipeline( + name="p", + tasks=[ + _notebook("writer", writes=[DataAsset(signature="shared_ds")]), + _notebook("reader", reads=[DataAsset(signature="shared_ds")]), + ], + ) + + edges = build_data_edges(pipeline) + + assert len(edges) == 1 + assert edges[0].match_kind == "signature" + assert edges[0].match_key == "shared_ds" + assert edges[0].identity is None + + +def test_data_edges_distinct_identities_do_not_fall_back_to_signature(): + """Two resolved-but-different identities never manufacture a signature edge (#36).""" + pipeline = Pipeline( + name="p", + tasks=[ + _notebook("writer", writes=[DataAsset(signature="shared", identity="a.first")]), + _notebook("reader", reads=[DataAsset(signature="shared", identity="b.second")]), + ], + ) + + assert build_data_edges(pipeline) == [] + + +def test_data_edges_fan_out_one_writer_many_readers(): + """One producer handing off to several consumers yields one edge each.""" + pipeline = Pipeline( + name="p", + tasks=[ + _notebook("writer", writes=[DataAsset(signature="ds", identity="x.y")]), + _notebook("reader_one", reads=[DataAsset(signature="ds", identity="x.y")]), + _notebook("reader_two", reads=[DataAsset(signature="ds", identity="x.y")]), + ], + ) + + edges = build_data_edges(pipeline) + + assert sorted(edge.target_task_key for edge in edges) == ["reader_one", "reader_two"] + assert all(edge.source_task_key == "writer" for edge in edges) + + +def test_data_edges_no_self_edge(): + """An activity that both writes and reads the same asset does not edge to itself.""" + pipeline = Pipeline( + name="p", + tasks=[ + _notebook( + "roundtrip", + writes=[DataAsset(signature="ds", identity="x.y")], + reads=[DataAsset(signature="ds", identity="x.y")], + ) + ], + ) + + assert build_data_edges(pipeline) == [] + + +def test_data_edges_no_duplicate_from_repeated_asset(): + """A producer listing the same asset twice still yields a single edge.""" + pipeline = Pipeline( + name="p", + tasks=[ + _notebook( + "writer", + writes=[DataAsset(signature="ds", identity="x.y"), DataAsset(signature="ds", identity="x.y")], + ), + _notebook("reader", reads=[DataAsset(signature="ds", identity="x.y")]), + ], + ) + + edges = build_data_edges(pipeline) + + assert len(edges) == 1 + + +def test_data_edges_nested_switch_recursion(): + """A producer buried in a Switch case hands off to a top-level consumer.""" + pipeline = Pipeline( + name="p", + tasks=[ + SwitchActivity( + name="sw", + task_key="sw", + on_expression="@e", + cases=[ + SwitchCase( + value="one", + activities=[_notebook("writer", writes=[DataAsset(signature="ds", identity="x.y")])], + ) + ], + default_activities=[], + ), + _notebook("reader", reads=[DataAsset(signature="ds", identity="x.y")]), + ], + ) + + edges = build_data_edges(pipeline) + + assert len(edges) == 1 + assert edges[0].source_task_key == "writer" + assert edges[0].target_task_key == "reader" + + +# --------------------------------------------------------------------------- # +# Motif annotations + composition + purity +# --------------------------------------------------------------------------- # + + +def test_build_motif_annotations_groups_members_by_tag(): + """A MotifActivity plus tagged members become one annotation over their task keys.""" + pipeline = Pipeline( + name="p", + tasks=[ + MotifActivity( + name="motif", + task_key="motif_auto_loader", + motif_id="auto_loader", + display_name="Auto Loader", + databricks_replacement="auto_loader", + matched_activity_names=["Copy A", "Copy B"], + confidence_notes=["matched on file source"], + ), + _notebook("member", motif_id="auto_loader"), + _notebook("unrelated"), + ], + ) + + annotations = build_motif_annotations(pipeline) + + assert len(annotations) == 1 + assert annotations[0].motif_id == "auto_loader" + assert annotations[0].member_task_keys == ["motif_auto_loader", "member"] + assert annotations[0].display_name == "Auto Loader" + assert annotations[0].databricks_replacement == "auto_loader" + + +def test_build_motif_annotations_carries_the_source_type_hint(): + """The detector's source classification survives into the IR lineage annotation.""" + pipeline = Pipeline( + name="p", + tasks=[ + MotifActivity( + name="motif", + task_key="motif_bulk_copy", + motif_id="metadata_driven_bulk_copy", + display_name="Bulk copy", + databricks_replacement="lakeflow_connect", + matched_activity_names=["Lookup", "ForEach"], + source_type_hint="database", + ), + ], + ) + + (annotation,) = build_motif_annotations(pipeline) + + assert annotation.source_type_hint == "database" + + +def test_with_lineage_is_pure(): + """with_lineage returns a new pipeline and never mutates the input.""" + pipeline = Pipeline(name="p", tasks=[_notebook("n")]) + lineage = build_lineage(pipeline) + + updated = with_lineage(pipeline, lineage) + + assert pipeline.lineage is None + assert updated is not pipeline + assert updated.lineage is lineage + + +# --------------------------------------------------------------------------- # +# ir_serde round-trips +# --------------------------------------------------------------------------- # + + +def test_serde_round_trip_new_activity_fields_and_lineage_block(): + """data_reads/data_writes/motif_id and the lineage block survive JSON round-trip.""" + pipeline = Pipeline( + name="p", + tasks=[ + _notebook( + "writer", + writes=[DataAsset(signature="ds_out", identity="curated.orders", asset_type="table")], + motif_id="auto_loader", + ), + _notebook( + "reader", + reads=[ + DataAsset( + signature="ds_in", + identity="curated.orders", + asset_type="table", + properties={"format": "delta"}, + ) + ], + ), + WaitActivity(name="pause", task_key="pause", wait_time_seconds=5), + ], + lineage=Lineage( + control_edges=[ + ControlEdge( + source_workflow="p", + target_workflow="child", + via_task_key="writer", + wait_for_completion=True, + resolved=True, + ) + ], + data_edges=[ + DataEdge( + source_task_key="writer", + target_task_key="reader", + match_kind="identity", + match_key="curated.orders", + identity="curated.orders", + asset_type="table", + ) + ], + motifs=[ + MotifAnnotation( + motif_id="auto_loader", + member_task_keys=["writer"], + display_name="Auto Loader", + databricks_replacement="auto_loader", + notes=["note"], + source_type_hint="AzureBlobFSReadSettings", + ) + ], + ), + ) + + reloaded, _ = pipeline_dict_to_ir(json.loads(json.dumps(pipeline_to_dict(pipeline)))) + + writer = reloaded.tasks[0] + reader = reloaded.tasks[1] + assert writer.motif_id == "auto_loader" + assert writer.data_writes == [DataAsset(signature="ds_out", identity="curated.orders", asset_type="table")] + assert reader.data_reads == [ + DataAsset(signature="ds_in", identity="curated.orders", asset_type="table", properties={"format": "delta"}) + ] + # A task without lineage fields rehydrates to empty lists / None, never missing. + assert reloaded.tasks[2].data_reads == [] + assert reloaded.tasks[2].data_writes == [] + assert reloaded.tasks[2].motif_id is None + + assert reloaded.lineage == pipeline.lineage + + +def test_serde_round_trip_motif_activity_no_kwarg_collision_r1(): + """A MotifActivity round-trips without the R1 'multiple values for motif_id' TypeError.""" + pipeline = Pipeline( + name="p", + tasks=[ + MotifActivity( + name="motif", + task_key="motif_auto_loader", + motif_id="auto_loader", + display_name="Auto Loader", + databricks_replacement="auto_loader", + matched_activity_names=["Copy A", "Copy B"], + data_reads=[DataAsset(signature="src", identity="raw.src")], + ) + ], + ) + + reloaded, _ = pipeline_dict_to_ir(json.loads(json.dumps(pipeline_to_dict(pipeline)))) + + task = reloaded.tasks[0] + assert isinstance(task, MotifActivity) + assert task.motif_id == "auto_loader" + assert task.display_name == "Auto Loader" + assert task.matched_activity_names == ["Copy A", "Copy B"] + assert task.data_reads == [DataAsset(signature="src", identity="raw.src")] + + +def test_lineage_block_always_emits_lists_never_null(): + """An attached empty lineage serialises its edge collections as lists, not null.""" + pipeline = with_lineage(Pipeline(name="p", tasks=[_notebook("n")]), Lineage()) + + serialised = pipeline_to_dict(pipeline) + + assert serialised["lineage"] == {"control_edges": [], "data_edges": [], "motifs": []}