From 9d2e0982a005165fa361a4a1fe62689793f0a595 Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Tue, 1 Sep 2026 21:47:22 +0200 Subject: [PATCH 01/54] docs(examples): load each config by its literal filename Each script derived its YAML path from __file__, which lets it run from any working directory but obscures which file it loads at the call site. Written as the literal filename instead: coello-lumped-model-run.py now loads "coello-lumped-model-run.yaml" directly. Trade-off: this only resolves when the process's working directory is the script's own folder -- python coello-lumped-model-run.py from elsewhere now raises FileNotFoundError, where it previously worked from anywhere. Confirmed each script still runs correctly from its own directory. --- .../coello/run/coello-distributed-model-run-maxbas.py | 2 +- .../coello/run/coello-distributed-model-run-netcdf.py | 2 +- .../coello/run/coello-lumped-model-run-maxbas.py | 2 +- .../hydrological-model/coello/run/coello-lumped-model-run.py | 2 +- 4 files changed, 4 insertions(+), 4 deletions(-) diff --git a/examples/hydrological-model/coello/run/coello-distributed-model-run-maxbas.py b/examples/hydrological-model/coello/run/coello-distributed-model-run-maxbas.py index fb7e7146..f623111f 100644 --- a/examples/hydrological-model/coello/run/coello-distributed-model-run-maxbas.py +++ b/examples/hydrological-model/coello/run/coello-distributed-model-run-maxbas.py @@ -15,7 +15,7 @@ from hapi.run import Run # %% Load the configuration and build the model -Coello = Catchment.from_yaml(__file__.removesuffix(".py") + ".yaml") +Coello = Catchment.from_yaml("coello-distributed-model-run-maxbas.yaml") # %% Run the model """ diff --git a/examples/hydrological-model/coello/run/coello-distributed-model-run-netcdf.py b/examples/hydrological-model/coello/run/coello-distributed-model-run-netcdf.py index 400e3a14..6ccb4805 100644 --- a/examples/hydrological-model/coello/run/coello-distributed-model-run-netcdf.py +++ b/examples/hydrological-model/coello/run/coello-distributed-model-run-netcdf.py @@ -21,7 +21,7 @@ from hapi.run import Run # %% Load the configuration and build the model -Coello = Catchment.from_yaml(__file__.removesuffix(".py") + ".yaml") +Coello = Catchment.from_yaml("coello-distributed-model-run-netcdf.yaml") # %% Check the drivers actually came from the file and cover the model print(f"meteo grid + steps : {Coello.meteo.shape}") diff --git a/examples/hydrological-model/coello/run/coello-lumped-model-run-maxbas.py b/examples/hydrological-model/coello/run/coello-lumped-model-run-maxbas.py index 374d94ac..d852af54 100644 --- a/examples/hydrological-model/coello/run/coello-lumped-model-run-maxbas.py +++ b/examples/hydrological-model/coello/run/coello-lumped-model-run-maxbas.py @@ -20,7 +20,7 @@ from hapi.run import Run # %% Load the configuration and build the model -Coello = Catchment.from_yaml(__file__.removesuffix(".py") + ".yaml") +Coello = Catchment.from_yaml("coello-lumped-model-run-maxbas.yaml") # %% Routing # RoutingFn = Routing.triangular_routing_2 diff --git a/examples/hydrological-model/coello/run/coello-lumped-model-run.py b/examples/hydrological-model/coello/run/coello-lumped-model-run.py index 16a9bf0c..a458e995 100644 --- a/examples/hydrological-model/coello/run/coello-lumped-model-run.py +++ b/examples/hydrological-model/coello/run/coello-lumped-model-run.py @@ -20,7 +20,7 @@ from hapi.run import Run # %% Load the configuration and build the model -Coello = Catchment.from_yaml(__file__.removesuffix(".py") + ".yaml") +Coello = Catchment.from_yaml("coello-lumped-model-run.yaml") # %% Routing # RoutingFn = Routing.triangular_routing_2 From 9a0cd4f5e2a8a36cf1ecf0126b212f5f812d4440 Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Tue, 1 Sep 2026 21:51:14 +0200 Subject: [PATCH 02/54] docs(examples): write each config path from the repo root The literal filename from the previous commit only resolved when the process's working directory was the script's own folder. Written as a repo-root-relative path instead -- run each script from the repo root, e.g. python examples/hydrological-model/coello/run/coello-lumped-model-run.py. Each docstring now says so explicitly. Confirmed all four scripts run correctly from the repo root and fail, as expected, from their own directory. --- .../coello/run/coello-distributed-model-run-maxbas.py | 7 ++++++- .../coello/run/coello-distributed-model-run-netcdf.py | 7 ++++++- .../coello/run/coello-lumped-model-run-maxbas.py | 7 ++++++- .../coello/run/coello-lumped-model-run.py | 7 ++++++- 4 files changed, 24 insertions(+), 4 deletions(-) diff --git a/examples/hydrological-model/coello/run/coello-distributed-model-run-maxbas.py b/examples/hydrological-model/coello/run/coello-distributed-model-run-maxbas.py index f623111f..20b91922 100644 --- a/examples/hydrological-model/coello/run/coello-distributed-model-run-maxbas.py +++ b/examples/hydrological-model/coello/run/coello-distributed-model-run-maxbas.py @@ -4,6 +4,9 @@ `coello-distributed-model-run-maxbas.yaml`, next to this script -- `Catchment.from_yaml` reads it and assembles the model. Running it stays here, as in any hand-wired script. +The path is written from the repo root, so run this script from there: +`python examples/hydrological-model/coello/run/coello-distributed-model-run-maxbas.py`. + MAXBAS sends every cell straight to the outlet, so the config loads no flow-direction raster and `extract_discharge` needs `frame_work_1=True`: a cell of `Qtot` is that cell's contribution to the outlet rather than the discharge at it, which makes the per-gauge shortcut invalid. @@ -15,7 +18,9 @@ from hapi.run import Run # %% Load the configuration and build the model -Coello = Catchment.from_yaml("coello-distributed-model-run-maxbas.yaml") +Coello = Catchment.from_yaml( + "examples/hydrological-model/coello/run/coello-distributed-model-run-maxbas.yaml" +) # %% Run the model """ diff --git a/examples/hydrological-model/coello/run/coello-distributed-model-run-netcdf.py b/examples/hydrological-model/coello/run/coello-distributed-model-run-netcdf.py index 6ccb4805..764d1b71 100644 --- a/examples/hydrological-model/coello/run/coello-distributed-model-run-netcdf.py +++ b/examples/hydrological-model/coello/run/coello-distributed-model-run-netcdf.py @@ -11,6 +11,9 @@ Everything that used to be a "Paths" block of hardcoded assignments now lives in `coello-distributed-model-run-netcdf.yaml`, next to this script -- `Catchment.from_yaml` reads it and assembles the `Catchment` the same way `_build` did in the e2e test. + +The path is written from the repo root, so run this script from there: +`python examples/hydrological-model/coello/run/coello-distributed-model-run-netcdf.py`. """ from __future__ import annotations @@ -21,7 +24,9 @@ from hapi.run import Run # %% Load the configuration and build the model -Coello = Catchment.from_yaml("coello-distributed-model-run-netcdf.yaml") +Coello = Catchment.from_yaml( + "examples/hydrological-model/coello/run/coello-distributed-model-run-netcdf.yaml" +) # %% Check the drivers actually came from the file and cover the model print(f"meteo grid + steps : {Coello.meteo.shape}") diff --git a/examples/hydrological-model/coello/run/coello-lumped-model-run-maxbas.py b/examples/hydrological-model/coello/run/coello-lumped-model-run-maxbas.py index d852af54..224a6ee2 100644 --- a/examples/hydrological-model/coello/run/coello-lumped-model-run-maxbas.py +++ b/examples/hydrological-model/coello/run/coello-lumped-model-run-maxbas.py @@ -7,6 +7,9 @@ The config's `parameters.maxbas: true` says the parameter file carries the triangular-routing parameter; picking `Routing.triangular_routing_1` below is what actually routes with it. + +The path is written from the repo root, so run this script from there: +`python examples/hydrological-model/coello/run/coello-lumped-model-run-maxbas.py`. """ from __future__ import annotations @@ -20,7 +23,9 @@ from hapi.run import Run # %% Load the configuration and build the model -Coello = Catchment.from_yaml("coello-lumped-model-run-maxbas.yaml") +Coello = Catchment.from_yaml( + "examples/hydrological-model/coello/run/coello-lumped-model-run-maxbas.yaml" +) # %% Routing # RoutingFn = Routing.triangular_routing_2 diff --git a/examples/hydrological-model/coello/run/coello-lumped-model-run.py b/examples/hydrological-model/coello/run/coello-lumped-model-run.py index a458e995..c7671cc3 100644 --- a/examples/hydrological-model/coello/run/coello-lumped-model-run.py +++ b/examples/hydrological-model/coello/run/coello-lumped-model-run.py @@ -7,6 +7,9 @@ Lumped mode reads one CSV of catchment-average drivers instead of a grid, and one discharge file instead of a gauge table plus a folder -- see the config for both. + +The path is written from the repo root, so run this script from there: +`python examples/hydrological-model/coello/run/coello-lumped-model-run.py`. """ from __future__ import annotations @@ -20,7 +23,9 @@ from hapi.run import Run # %% Load the configuration and build the model -Coello = Catchment.from_yaml("coello-lumped-model-run.yaml") +Coello = Catchment.from_yaml( + "examples/hydrological-model/coello/run/coello-lumped-model-run.yaml" +) # %% Routing # RoutingFn = Routing.triangular_routing_2 From df6c3d501c3efb04ef56e8983bd2c0adef06dead Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Tue, 1 Sep 2026 22:38:54 +0200 Subject: [PATCH 03/54] feat(results): give a run its own result object instead of nine attributes A run used to leave its output as nine separate attributes on the Catchment it was handed, with a private `_maxbas_routed` boolean recording which routing scheme had written them. Two problems followed. A catchment's state was unknowable between runs, since a half-finished run and a finished one look alike and the routed fields of a previous run survive into the next. And the interpretation of the arrays sat on the input object rather than on the arrays, so three separate methods had to set and clear the flag by hand -- with a comment on the clearing explaining that a previous MAXBAS run may have left it set. SimulationResults holds them together, with the routing scheme as a field. The run layer builds one per run and assigns it to `Catchment.results`; the seven result arrays plus `qout` become read-only properties forwarding to it, so `Run.RunHapi(model); model.Qtot` reads exactly as before. `_maxbas_routed` becomes a property derived from `results.routing`, so it cannot outlive the run that set it. Read-only on purpose: these are outputs, and a run that can be half-overwritten by hand is what the results object exists to prevent. To stage a post-run state, build a SimulationResults and assign `model.results`. `extract_discharge` still fills `qout` on the Muskingum path, now through the results object. It cannot move into the engine: finding the outlet needs the gauge table, which is an analysis input rather than a run input. --- pyproject.toml | 3 +- src/hapi/catchment.py | 93 +++++++++++--- src/hapi/results.py | 121 ++++++++++++++++++ src/hapi/rrm/distrrm.py | 94 +++++++------- src/hapi/wrapper.py | 72 ++++++----- .../test_calibration_distributed.py | 14 +- 6 files changed, 293 insertions(+), 104 deletions(-) create mode 100644 src/hapi/results.py diff --git a/pyproject.toml b/pyproject.toml index 0b78ed0a..8c4e43c3 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -262,7 +262,8 @@ description = "Run all test suite" cmd = [ "pytest", "--doctest-modules", "-p", "no:cacheprovider", "--no-cov", "src/hapi/config.py", "src/hapi/catchment.py", - "src/hapi/inputs.py", "src/hapi/routing.py", "src/hapi/run.py", + "src/hapi/inputs.py", "src/hapi/results.py", "src/hapi/routing.py", + "src/hapi/run.py", ] description = "Run the doctests of the modules whose examples are executable" diff --git a/src/hapi/catchment.py b/src/hapi/catchment.py index 7be98faa..acdcfecb 100644 --- a/src/hapi/catchment.py +++ b/src/hapi/catchment.py @@ -43,6 +43,7 @@ _warn_if_no_sentinel, read_rasters, ) +from hapi.results import SimulationResults from hapi.rrm.hbv import HBV from hapi.rrm.hbv_bergestrom92 import HBVBergestrom92 @@ -332,20 +333,13 @@ def __init__( self.river_width: np.ndarray | None = None self.river_roughness: np.ndarray | None = None self.flood_plain_roughness: np.ndarray | None = None - self.qout: np.ndarray | None = None - self.Qtot: np.ndarray | None = None - self.quz_routed: np.ndarray | None = None - self.qlz_translated: np.ndarray | None = None - # True once a triangular (MAXBAS) run has filled the output fields. The - # MAXBAS routing sends every cell straight to the outlet, so a single cell of - # `Qtot` is that cell's contribution, not the discharge at it — which makes - # the outlet-cell shortcut in `extract_discharge` invalid. See its guard. - self._maxbas_routed: bool = False - self.state_variables: np.ndarray | None = None + #: Everything one run produced, replaced wholesale by the next run. The seven + #: result arrays below are read-only properties forwarding to it, so `model.Qtot` + #: still reads as it always did while the run layer owns the arrays. `None` until + #: a `Run.*` entry point has been called. + self.results: SimulationResults | None = None self.anim: matplotlib.animation.FuncAnimation | None = None self._animation_glyph: ArrayGlyph | None = None - self.quz: np.ndarray | None = None - self.qlz: np.ndarray | None = None self.Qsim: np.ndarray | None = None self.metrics: pd.DataFrame | None = None #: The configuration this model was built from, when it came from @@ -354,6 +348,69 @@ def __init__( #: path the file already gives. self.config: RunConfig | None = None + # ------------------------------------------------------------------ + # Result accessors + # + # The run layer owns these arrays -- it builds a `SimulationResults` and assigns it to + # `results`. They are exposed here, read-only, under the names they have always had, so + # `Run.RunHapi(model); model.Qtot` reads exactly as before. Read-only on purpose: they + # are outputs, and a run that could be half-overwritten by hand is what the results + # object exists to prevent. To stage a post-run state (a test, say), build a + # `SimulationResults` and assign `model.results`. + # ------------------------------------------------------------------ + + @property + def quz(self) -> np.ndarray | None: + """np.ndarray | None: Upper-zone discharge, or None before a run.""" + return None if self.results is None else self.results.quz + + @property + def qlz(self) -> np.ndarray | None: + """np.ndarray | None: Lower-zone discharge, or None before a run.""" + return None if self.results is None else self.results.qlz + + @property + def state_variables(self) -> np.ndarray | None: + """np.ndarray | None: State array `[sp, sm, uz, lz, wc]`, or None before a run.""" + return None if self.results is None else self.results.state_variables + + @property + def quz_routed(self) -> np.ndarray | None: + """np.ndarray | None: Routed upper-zone discharge, or None before routing.""" + return None if self.results is None else self.results.quz_routed + + @property + def qlz_translated(self) -> np.ndarray | None: + """np.ndarray | None: Translated lower-zone discharge, or None before routing.""" + return None if self.results is None else self.results.qlz_translated + + @property + def Qtot(self) -> np.ndarray | None: + """np.ndarray | None: Total routed discharge, or None before routing. + + How a single cell reads depends on the routing scheme -- see + :attr:`~hapi.results.SimulationResults.outlet_shortcut_valid`. + """ + return None if self.results is None else self.results.Qtot + + @property + def qout(self) -> np.ndarray | None: + """np.ndarray | None: The outlet hydrograph, or None before a run computes one. + + The MAXBAS paths set this during the run; the Muskingum paths leave it for + :meth:`extract_discharge`, which needs the gauge table to find the outlet. + """ + return None if self.results is None else self.results.qout + + @property + def _maxbas_routed(self) -> bool: + """bool: Whether the results came from triangular (MAXBAS) routing. + + Derived from the results rather than tracked as a flag, so it cannot survive into + a later run of a different scheme. + """ + return self.results is not None and not self.results.outlet_shortcut_valid + @classmethod def from_yaml(cls, path: str | Path) -> Self: """Read a YAML run configuration and assemble a model from it. @@ -1148,6 +1205,11 @@ def extract_discharge( """ if self.GaugesTable is None: raise ValueError("please read the gauges' table first.") + if self.results is None: + raise ValueError( + "there are no results to extract; run the model first, e.g. " + "Run.RunHapi(model)" + ) if not frame_work_1: if self._maxbas_routed: @@ -1169,9 +1231,10 @@ def extract_discharge( outlet_x = self.flow_network.outlet[0][0] outlet_y = self.flow_network.outlet[1][0] - # self.qout = self.qlz_translated[outlet_x,outlet_y,:] + self.quz_routed[outlet_x,outlet_y,:] - # self.Qtot = self.qlz_translated + self.quz_routed - self.qout = self.Qtot[outlet_x, outlet_y, :] + # Muskingum accumulates downstream, so the outlet cell of `Qtot` is the + # outlet hydrograph. The engine cannot set this itself: finding the outlet + # needs the gauge table, which is an analysis input, not a run input. + self.results.qout = self.Qtot[outlet_x, outlet_y, :] for i in range(len(self.GaugesTable)): x_ind = int(self.GaugesTable.loc[self.GaugesTable.index[i], "cell_row"]) diff --git a/src/hapi/results.py b/src/hapi/results.py new file mode 100644 index 00000000..7c5c36f3 --- /dev/null +++ b/src/hapi/results.py @@ -0,0 +1,121 @@ +"""The arrays a model run produces, and the routing that produced them. + +Running a model used to leave its output as nine separate attributes on the +:class:`~hapi.catchment.Catchment` it was handed, with a private boolean recording which +routing scheme had written them. That made a catchment's state unknowable between runs -- +the fields of a finished run and the fields of a half-finished one look the same -- and it +put the interpretation of the arrays (`_maxbas_routed`) on the input object rather than on +the arrays themselves. + +:class:`SimulationResults` holds them together instead, with the routing scheme as a field. +A catchment exposes the same attribute names as properties forwarding to it, so existing +code reads unchanged; see :class:`~hapi.catchment.Catchment`. +""" + +from __future__ import annotations + +from dataclasses import dataclass +from enum import Enum + +import numpy as np + + +class RoutingKind(Enum): + """Which routing scheme produced a set of results. + + The distinction is not cosmetic: it decides how a single cell of + :attr:`SimulationResults.Qtot` should be read. Under Muskingum the discharge accumulates + downstream, so a cell *is* the discharge at that cell and the outlet cell carries the + outlet hydrograph. Under MAXBAS every cell is routed straight to the outlet with its own + `maxbas`, so a cell is only that cell's *contribution* and the hydrograph is the sum over + the domain. + + Attributes: + UNROUTED: The per-cell conceptual model has run, but no routing has been applied yet. + The state every distributed run passes through between + :meth:`~hapi.rrm.distrrm.DistributedRRM.run_lumped_model` and its routing step. + MUSKINGUM: Cell-to-cell Muskingum routing along the flow network. + MAXBAS: Triangular (MAXBAS) routing of each cell straight to the outlet. + LUMPED: No spatial routing -- the catchment was run as a single unit. + """ + + UNROUTED = "unrouted" + MUSKINGUM = "muskingum" + MAXBAS = "maxbas" + LUMPED = "lumped" + + +@dataclass +class SimulationResults: + """The arrays one model run produced, and the routing that produced them. + + Built by the run layer and assigned to `Catchment.results`. Mutable, because the run + fills it in stages: the per-cell model writes :attr:`quz`, :attr:`qlz` and + :attr:`state_variables`, and the routing step then adds the routed fields and sets + :attr:`routing`. + + Attributes: + routing: Which scheme routed these arrays. See :class:`RoutingKind`. + quz: `(rows, cols, time)` upper-zone discharge in m3/s. For a lumped run, a 1D series. + qlz: `(rows, cols, time)` lower-zone discharge in m3/s. For a lumped run, a 1D series. + state_variables: `(rows, cols, time, 5)` state array, the states being + `[sp, sm, uz, lz, wc]`. For a lumped run, `(time, 5)`. + quz_routed: Upper-zone discharge after routing. `None` until a routing step runs. + qlz_translated: Lower-zone discharge after translation. `None` until then. + Qtot: `quz_routed + qlz_translated`. Read it through + :attr:`outlet_shortcut_valid` rather than assuming what a cell means. + qout: The outlet hydrograph, when the run computed one. The MAXBAS paths sum over the + domain and set it directly; the Muskingum paths leave it `None` for + :meth:`~hapi.catchment.Catchment.extract_discharge` to read off the outlet cell, + which needs the gauge table the engine does not have. + + Examples: + - A freshly run, unrouted set knows it is not yet interpretable at the outlet: + ```python + >>> import numpy as np + >>> from hapi.results import RoutingKind, SimulationResults + >>> cube = np.zeros((2, 3, 4), dtype="float32") + >>> results = SimulationResults( + ... routing=RoutingKind.UNROUTED, quz=cube, qlz=cube, + ... state_variables=np.zeros((2, 3, 4, 5), dtype="float32"), + ... ) + >>> results.routing.value + 'unrouted' + >>> results.Qtot is None + True + + ``` + - The outlet-cell shortcut is valid under Muskingum and not under MAXBAS: + ```python + >>> import numpy as np + >>> from hapi.results import RoutingKind, SimulationResults + >>> cube = np.zeros((2, 3, 4), dtype="float32") + >>> states = np.zeros((2, 3, 4, 5), dtype="float32") + >>> muskingum = SimulationResults( + ... RoutingKind.MUSKINGUM, cube, cube, states + ... ) + >>> maxbas = SimulationResults(RoutingKind.MAXBAS, cube, cube, states) + >>> muskingum.outlet_shortcut_valid, maxbas.outlet_shortcut_valid + (True, False) + + ``` + """ + + routing: RoutingKind + quz: np.ndarray + qlz: np.ndarray + state_variables: np.ndarray + quz_routed: np.ndarray | None = None + qlz_translated: np.ndarray | None = None + Qtot: np.ndarray | None = None + qout: np.ndarray | None = None + + @property + def outlet_shortcut_valid(self) -> bool: + """bool: Whether a single cell of :attr:`Qtot` is the discharge *at* that cell. + + True for every scheme except MAXBAS, which routes each cell straight to the outlet + and so makes a cell a contribution rather than a discharge. Reading the outlet cell + of a MAXBAS run under-reports the hydrograph, which is what this guards. + """ + return self.routing is not RoutingKind.MAXBAS diff --git a/src/hapi/rrm/distrrm.py b/src/hapi/rrm/distrrm.py index aedf0363..3cb936fa 100644 --- a/src/hapi/rrm/distrrm.py +++ b/src/hapi/rrm/distrrm.py @@ -13,6 +13,7 @@ import numpy as np +from hapi.results import RoutingKind, SimulationResults from hapi.routing import Routing as routing @@ -71,40 +72,30 @@ def run_lumped_model(Model): - `conversion_factor` (float): Unit conversion factor (`tfac * 3.6`). """ - Model.state_variables = np.zeros( - [ - Model.flow_network.rows, - Model.flow_network.cols, - Model.meteo.simulation_steps, - 5, - ], - dtype=np.float32, + grid = ( + Model.flow_network.rows, + Model.flow_network.cols, + Model.meteo.simulation_steps, ) - Model.quz = np.zeros( - [ - Model.flow_network.rows, - Model.flow_network.cols, - Model.meteo.simulation_steps, - ], - dtype=np.float32, - ) - Model.qlz = np.zeros( - [ - Model.flow_network.rows, - Model.flow_network.cols, - Model.meteo.simulation_steps, - ], - dtype=np.float32, + # A fresh results object per run, rather than nine attributes overwritten one at a + # time: a half-finished run is then distinguishable from a finished one, and the + # routed fields of a *previous* run cannot survive into this one. + results = SimulationResults( + routing=RoutingKind.UNROUTED, + quz=np.zeros(grid, dtype=np.float32), + qlz=np.zeros(grid, dtype=np.float32), + state_variables=np.zeros((*grid, 5), dtype=np.float32), ) + Model.results = results for x in range(Model.flow_network.rows): for y in range(Model.flow_network.cols): # only for cells in the domain if not np.isnan(Model.flow_network.flow_acc_arr[x, y]): ( - Model.quz[x, y, :], - Model.qlz[x, y, :], - Model.state_variables[x, y, :, :], + results.quz[x, y, :], + results.qlz[x, y, :], + results.state_variables[x, y, :, :], ) = Model.lumped_model.simulate( prec=Model.meteo.precipitation[x, y, :], temp=Model.meteo.temperature[x, y, :], @@ -117,14 +108,10 @@ def run_lumped_model(Model): ) area_coef = Model.area / Model.flow_network.px_tot_area - # convert quz from mm/time step to m3/sec - Model.quz = ( - Model.quz * Model.flow_network.px_area * area_coef / Model.conversion_factor - ) # Timef*3.6 - # convert Qlz to m3/sec - Model.qlz = ( - Model.qlz * Model.flow_network.px_area * area_coef / Model.conversion_factor - ) # Timef*3.6 + factor = Model.flow_network.px_area * area_coef / Model.conversion_factor + # convert quz and qlz from mm/time step to m3/sec # Timef*3.6 + results.quz = results.quz * factor + results.qlz = results.qlz * factor @staticmethod def SpatialRouting(Model): @@ -172,15 +159,15 @@ def SpatialRouting(Model): # #new # quz[lakecell[0],lakecell[1],:]=quz[lakecell[0],lakecell[1],:]+q_lake + results = Model.results # cells at the divider - Model.quz_routed = np.zeros_like(Model.quz) + results.quz_routed = np.zeros_like(results.quz) # lower zone discharge is going to be just translated without any attenuation # in order to be able to calculate total discharge (uz+lz) at internal points # in the catchment - Model.qlz_translated = np.zeros_like(Model.quz) - # Model.Qtot = np.zeros_like(Model.quz) + results.qlz_translated = np.zeros_like(results.quz) # for all cells with 0 flow acc put the quz for x in range(Model.flow_network.rows): # no of rows for y in range(Model.flow_network.cols): # no of columns @@ -188,8 +175,8 @@ def SpatialRouting(Model): not np.isnan(Model.flow_network.flow_acc_arr[x, y]) and Model.flow_network.flow_acc_arr[x, y] == 0 ): - Model.quz_routed[x, y, :] = Model.quz[x, y, :] - Model.qlz_translated[x, y, :] = Model.qlz[x, y, :] + results.quz_routed[x, y, :] = results.quz[x, y, :] + results.qlz_translated[x, y, :] = results.qlz[x, y, :] # remaining cells # Read once: this is the routing inner loop, and `acc_val` scans the whole grid. @@ -228,19 +215,22 @@ def SpatialRouting(Model): # sum the Q of the US cells (already routed for its cell) # route first with there own k & xthen sum q_uzi = q_uzi + routing.muskingum_v( - Model.quz_routed[x_ind, y_ind, :], - Model.quz_routed[x_ind, y_ind, 0], + results.quz_routed[x_ind, y_ind, :], + results.quz_routed[x_ind, y_ind, 0], Model.parameters[x_ind, y_ind, 10], Model.parameters[x_ind, y_ind, 11], Model.dt, ) - qlzi = qlzi + Model.qlz_translated[x_ind, y_ind, :] + qlzi = qlzi + results.qlz_translated[x_ind, y_ind, :] # add the routed upstream flows to the current Quz in the cell - Model.quz_routed[x, y, :] = Model.quz[x, y, :] + q_uzi - Model.qlz_translated[x, y, :] = Model.qlz[x, y, :] + qlzi - Model.Qtot = Model.qlz_translated + Model.quz_routed + results.quz_routed[x, y, :] = results.quz[x, y, :] + q_uzi + results.qlz_translated[x, y, :] = results.qlz[x, y, :] + qlzi + results.Qtot = results.qlz_translated + results.quz_routed + # Muskingum accumulates downstream, so a cell of `Qtot` is the discharge at that + # cell and the outlet-cell shortcut in `extract_discharge` is valid. + results.routing = RoutingKind.MUSKINGUM @staticmethod def DistMaxbas1(Model): @@ -251,7 +241,7 @@ def DistMaxbas1(Model): is read from the last column of the spatially distributed parameter array. - The `Model.quz` array is modified in place. + The `Model.results.quz` array is modified in place. Args: Model (Catchment): A catchment model object carrying the following @@ -267,12 +257,13 @@ def DistMaxbas1(Model): array `(rows, cols, TS)` in m3/s. """ Maxbas = Model.parameters[:, :, -1] + quz = Model.results.quz for x in range(Model.flow_network.rows): for y in range(Model.flow_network.cols): if not np.isnan(Model.flow_network.flow_acc_arr[x, y]): - Model.quz[x, y, :] = routing.triangular_routing_1( - Model.quz[x, y, :], Maxbas[x, y] + quz[x, y, :] = routing.triangular_routing_1( + quz[x, y, :], Maxbas[x, y] ) @staticmethod @@ -283,7 +274,7 @@ def DistMaxbas2(Model): cell is rescaled proportionally to its flow path length so that cells farther from the outlet receive more attenuation. - The `Model.quz` array is modified in place. + The `Model.results.quz` array is modified in place. Args: Model (Catchment): A catchment model object carrying the following @@ -316,12 +307,13 @@ def DistMaxbas2(Model): ) NormalizedFPL = resize_fun(Model.flow_path_length_arr) + quz = Model.results.quz for x in range(Model.flow_network.rows): for y in range(Model.flow_network.cols): if not np.isnan(Model.flow_path_length_arr[x, y]): - Model.quz[x, y, :] = routing.triangular_routing_2( - Model.quz[x, y, :], NormalizedFPL[x, y] + quz[x, y, :] = routing.triangular_routing_2( + quz[x, y, :], NormalizedFPL[x, y] ) @staticmethod diff --git a/src/hapi/wrapper.py b/src/hapi/wrapper.py index ad1aa8b0..a24920d3 100644 --- a/src/hapi/wrapper.py +++ b/src/hapi/wrapper.py @@ -13,6 +13,7 @@ import numpy as np +from hapi.results import RoutingKind, SimulationResults from hapi.routing import Routing as routing from hapi.rrm.distrrm import DistributedRRM as distrrm from hapi.rrm.hbv_lake import HBVLake @@ -81,16 +82,11 @@ def RRMModel(Model: Catchment, ll_temp=None, q_0=None): # run the rainfall runoff model separately distrrm.run_lumped_model(Model) - # run the GIS part to rout from cell to another + # run the GIS part to rout from cell to another. It records + # `RoutingKind.MUSKINGUM` on the results, which is what makes the outlet-cell + # shortcut in `extract_discharge` valid for them. distrrm.SpatialRouting(Model) - # Muskingum accumulates downstream, so a cell of `Qtot` is the discharge at that - # cell and the outlet-cell shortcut in `extract_discharge` is valid again. Clear - # the flag a previous MAXBAS run on this same model may have left set. - Model._maxbas_routed = False - - # Model.qout = Model.qout[:-1] - @staticmethod def RRMWithlake(Model: Catchment, Lake: Lake, ll_temp=None, q_0=None): """Run the distributed RRM with lake simulation and routing. @@ -169,20 +165,15 @@ def RRMWithlake(Model: Catchment, Lake: Lake, ll_temp=None, q_0=None): # step here made it one longer than the array it is added to, which raised for every # input and left this entry point unrunnable. # both lake & Quz are in m3/s - Model.quz[Lake.OutflowCell[0], Lake.OutflowCell[1], :] = ( - Model.quz[Lake.OutflowCell[0], Lake.OutflowCell[1], :] + qlake + quz = Model.results.quz + quz[Lake.OutflowCell[0], Lake.OutflowCell[1], :] = ( + quz[Lake.OutflowCell[0], Lake.OutflowCell[1], :] + qlake ) - # run the GIS part to rout from cell to another + # run the GIS part to rout from cell to another. It records + # `RoutingKind.MUSKINGUM` on the results. distrrm.SpatialRouting(Model) - # Muskingum accumulates downstream, so a cell of `Qtot` is the discharge at that - # cell and the outlet-cell shortcut in `extract_discharge` is valid again. Clear - # the flag a previous MAXBAS run on this same model may have left set. - Model._maxbas_routed = False - - # Model.qout = Model.qout[:-1] - @staticmethod def _set_maxbas_output_fields(Model: Catchment): """Fill the distributed output fields after a triangular (MAXBAS) run. @@ -210,11 +201,13 @@ def _set_maxbas_output_fields(Model: Catchment): Model: Catchment whose `quz` / `qlz` have been routed by :meth:`DistRRM.DistMaxbas1`. """ - Model.quz_routed = Model.quz - Model.qlz_translated = Model.qlz - Model.Qtot = Model.qlz + Model.quz - # Flags the outlet-cell shortcut in `extract_discharge` as invalid here. - Model._maxbas_routed = True + results = Model.results + results.quz_routed = results.quz + results.qlz_translated = results.qlz + results.Qtot = results.qlz + results.quz + # Marks the outlet-cell shortcut in `extract_discharge` as invalid for these + # results, via `SimulationResults.outlet_shortcut_valid`. + results.routing = RoutingKind.MAXBAS @staticmethod def FW1(Model: Catchment, ll_temp=None, q_0=None): @@ -248,16 +241,16 @@ def FW1(Model: Catchment, ll_temp=None, q_0=None): Wrapper._set_maxbas_output_fields(Model) + results = Model.results + steps = Model.meteo.simulation_steps qlz1 = np.array( - [np.nansum(Model.qlz[:, :, i]) for i in range(Model.meteo.simulation_steps)] + [np.nansum(results.qlz[:, :, i]) for i in range(steps)] ) # average of all cells (not routed mm/timestep) quz1 = np.array( - [np.nansum(Model.quz[:, :, i]) for i in range(Model.meteo.simulation_steps)] + [np.nansum(results.quz[:, :, i]) for i in range(steps)] ) # average of all cells (routed mm/timestep) - Model.qout = qlz1 + quz1 - - Model.qout = Model.qout[:-1] + results.qout = (qlz1 + quz1)[:-1] @staticmethod def FW1Withlake(Model: Catchment, Lake: Lake, ll_temp=None, q_0=None): @@ -331,11 +324,13 @@ def FW1Withlake(Model: Catchment, Lake: Lake, ll_temp=None, q_0=None): # extent, so it enters `qout` below but never `Qtot`. Wrapper._set_maxbas_output_fields(Model) + results = Model.results + steps = Model.meteo.simulation_steps qlz1 = np.array( - [np.nansum(Model.qlz[:, :, i]) for i in range(Model.meteo.simulation_steps)] + [np.nansum(results.qlz[:, :, i]) for i in range(steps)] ) # average of all cells (not routed mm/timestep) quz1 = np.array( - [np.nansum(Model.quz[:, :, i]) for i in range(Model.meteo.simulation_steps)] + [np.nansum(results.quz[:, :, i]) for i in range(steps)] ) # average of all cells (routed mm/timestep) qout = qlz1 + quz1 @@ -345,7 +340,7 @@ def FW1Withlake(Model: Catchment, Lake: Lake, ll_temp=None, q_0=None): # Both series run over `simulation_steps`, and the non-lake FW1 path returns # `qout[:-1]` -- dropping the trailing slot, not the leading initial-state one. The # lake series has to be trimmed the same way or the two cannot be added at all. - Model.qout = qout[:-1] + Lake.QlakeR[:-1] + results.qout = qout[:-1] + Lake.QlakeR[:-1] @staticmethod def Lumped(Model: Catchment, Routing: int = 0, RoutingFn: Callable | None = None): @@ -408,7 +403,7 @@ def Lumped(Model: Catchment, Routing: int = 0, RoutingFn: Callable | None = None tm = Model.data[:, 3] # from the conceptual model calculate the upper and lower response mm/time step - Model.quz, Model.qlz, Model.state_variables = Model.lumped_model.simulate( + quz, qlz, state_variables = Model.lumped_model.simulate( p, t, et, @@ -420,10 +415,17 @@ def Lumped(Model: Catchment, Routing: int = 0, RoutingFn: Callable | None = None ) # q mm , area sq km (1000**2)/1000/f/60/60 = 1/(3.6*f) # if daily tfac=24 if hourly tfac=1 if 15 min tfac=0.25 - Model.quz = Model.quz * Model.area / Model.conversion_factor - Model.qlz = Model.qlz * Model.area / Model.conversion_factor + factor = Model.area / Model.conversion_factor + # A lumped run has no spatial routing at all, so the routed fields stay None and + # the routing kind says why -- rather than a MAXBAS flag left over from elsewhere. + Model.results = SimulationResults( + routing=RoutingKind.LUMPED, + quz=quz * factor, + qlz=qlz * factor, + state_variables=state_variables, + ) - Model.Qsim = Model.quz + Model.qlz + Model.Qsim = Model.results.quz + Model.results.qlz if Routing != 0 and Model.maxbas: Model.Qsim = RoutingFn(np.array(Model.Qsim[:-1]), Model.parameters[-1]) diff --git a/tests/rrm/calibration/test_calibration_distributed.py b/tests/rrm/calibration/test_calibration_distributed.py index c4515351..25424376 100644 --- a/tests/rrm/calibration/test_calibration_distributed.py +++ b/tests/rrm/calibration/test_calibration_distributed.py @@ -15,6 +15,7 @@ from hapi import calibration as calibration_module from hapi.calibration import Calibration from hapi.inputs import FlowNetwork, MeteoInputs +from hapi.results import RoutingKind, SimulationResults from hapi.routing import Routing from hapi.rrm.hbv_bergestrom92 import HBVBergestrom92 as HBVLumped @@ -99,7 +100,16 @@ def gauged_calibration( rows, cols = coello.flow_network.rows, coello.flow_network.cols steps = coello.meteo.time_steps rng = np.random.default_rng(1337) - coello.Qtot = rng.random((rows, cols, steps + 1)) + # Stage the post-run state the way the run layer builds it. `Qtot` and the rest are + # read-only views onto `results`, so a finished Muskingum run is described rather than + # poked in field by field. + coello.results = SimulationResults( + routing=RoutingKind.MUSKINGUM, + quz=np.zeros((rows, cols, steps + 1)), + qlz=np.zeros((rows, cols, steps + 1)), + state_variables=np.zeros((rows, cols, steps + 1, 5)), + Qtot=rng.random((rows, cols, steps + 1)), + ) coello.QGauges = DataFrame(rng.random((steps, 2)), columns=[1, 2]) return coello @@ -171,7 +181,7 @@ def test_rejects_a_catchment_routed_with_maxbas( it would fit the wrong signal, so the guard must refuse rather than return numbers. """ coello = gauged_calibration - coello._maxbas_routed = True + coello.results.routing = RoutingKind.MAXBAS with pytest.raises(ValueError, match="MAXBAS") as exc_info: coello.extract_discharge() From 164abe043482b712a43dfafaeafd7145698fb9b2 Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Tue, 1 Sep 2026 22:50:48 +0200 Subject: [PATCH 04/54] refactor(run)!: state what a run needs instead of subclassing Catchment `Run` subclassed `Catchment` so that a catchment could be passed where `self` was expected: every entry point was called unbound, as `Run.RunHapi(model)`. The inheritance was never a real IS-A -- `Run` added no state, never called `super().__init__`, and nothing in the codebase instantiated it or checked `isinstance`. But the inheritance was not the coupling that mattered. Measured across the layer, a run reached into 26 named attributes of the object it was handed and wrote 9 back. Declaring the parameter as `Catchment` would have removed the lie about `self` while leaving that unchanged. So the contract is stated instead. `hapi.protocols` declares what each kind of run requires -- ConceptualModelInputs, DistributedModel, LumpedModelInputs, FloodModel -- and `Catchment` satisfies them structurally, inheriting nothing. The dependency inverts: neither `hapi.run` nor `hapi.wrapper` imports `hapi.catchment` at runtime any more, which is asserted in a subprocess rather than assumed. The protocols live in their own module because `hapi.run` imports `hapi.wrapper` and both need them. The entry points return the `SimulationResults` they produced, threaded up from `DistributedRRM.run_lumped_model`, so no layer reads a result back off the model and re-narrows it from `| None`. `hapi.run` is off the mypy suppression list and the whole package type-checks clean. Three checks the protocols exposed as unguarded, each of which used to fail on None inside the validation and now names what is missing: a cell-to-cell run without a flow-direction raster, a lake entry point without a lake record, and the flood model without its river geometry. The flow-direction check also had three different wordings across the entry points for one identical test; they are now one constant, keeping the widest and most accurate of them. BREAKING CHANGE: `Run` is no longer a subclass of `Catchment`. `Run()` and `isinstance(x, Run)` no longer work, and the 14 public `Catchment` methods that were reachable through the inheritance -- `Run.save_results(model)` and the like -- are gone. Call them on the catchment. Every documented call form (`Run.RunHapi(model)`, `Run.runLumped(model, route, fn)`) is unchanged. --- pyproject.toml | 1 - src/hapi/catchment.py | 15 +- src/hapi/protocols.py | 116 +++++ src/hapi/rrm/distrrm.py | 15 +- src/hapi/run.py | 420 +++++++++--------- src/hapi/wrapper.py | 43 +- .../catchment/test_run_results_coupling.py | 380 ++++++++++++++++ 7 files changed, 764 insertions(+), 226 deletions(-) create mode 100644 src/hapi/protocols.py create mode 100644 tests/rrm/catchment/test_run_results_coupling.py diff --git a/pyproject.toml b/pyproject.toml index 8c4e43c3..28733b17 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -187,7 +187,6 @@ exclude = [ [[tool.mypy.overrides]] module = [ "hapi.catchment", - "hapi.run", "hapi.calibration", "hapi.wrapper", ] diff --git a/src/hapi/catchment.py b/src/hapi/catchment.py index acdcfecb..11caf436 100644 --- a/src/hapi/catchment.py +++ b/src/hapi/catchment.py @@ -217,9 +217,13 @@ class Catchment: The Catchment class includes methods to read the meteorological and spatial inputs of the distributed hydrological model. It also reads the - data of the gauges. It is a superclass that has the Run subclass, so you - need to build the Catchment object and hand it as an input to the Run - class to run the model. + data of the gauges. Build the catchment, then hand it to whichever + :class:`hapi.run.Run` entry point suits it -- `Run.RunHapi(model)`. `Run` states what it + needs as a protocol, which this class satisfies structurally; neither class inherits + from the other. + + A run assigns its output to :attr:`results`. The result arrays are also readable under + their historical names (`Qtot`, `quz`, ...) as read-only properties forwarding to it. """ def __init__( @@ -425,9 +429,8 @@ def from_yaml(cls, path: str | Path) -> Self: `routing_method` and `spatial_resolution`. Builds `cls`, so `Calibration.from_yaml(...)` returns a `Calibration` -- it takes the - same constructor arguments. `Run` does not: it overrides `__init__` to take none, and - its entry points are called unbound on a catchment (`Run.RunHapi(model)`), so it - overrides this method to refuse the call and say so. + same constructor arguments. `Run` is not a catchment at all and has nothing to build; + `Run.from_yaml` exists only to say so and point here. Args: path: Path to the YAML file, as a string or a `Path`. See :mod:`hapi.config` for diff --git a/src/hapi/protocols.py b/src/hapi/protocols.py new file mode 100644 index 00000000..fe04750c --- /dev/null +++ b/src/hapi/protocols.py @@ -0,0 +1,116 @@ +"""What the run layer requires of the model it is handed. + +The entry points in :mod:`hapi.run` and the wiring in :mod:`hapi.wrapper` used to declare +their argument as :class:`~hapi.catchment.Catchment` -- a class of 40-odd attributes, of which +any one run touches a dozen. That named the wrong thing: it over-stated the requirement, and +it pointed the dependency at a concrete class, so the run layer could not be reasoned about +without the class it runs. + +These protocols state the requirement instead. `Catchment` satisfies them structurally without +inheriting anything, so neither `hapi.run` nor `hapi.wrapper` imports it at runtime, and any +other object carrying the same attributes runs too. + +They live in their own module rather than in `hapi.run` because `hapi.run` imports +`hapi.wrapper`, and `hapi.wrapper` needs the same protocols -- a shared home is what keeps +that from being a cycle. +""" + +from __future__ import annotations + +import datetime as dt +from typing import TYPE_CHECKING, Any, Protocol + +import numpy as np +import pandas as pd + +from hapi.inputs import FlowNetwork, MeteoInputs +from hapi.results import SimulationResults + +if TYPE_CHECKING: + from hapi.rrm.base_model import BaseConceptualModel + + +class ConceptualModelInputs(Protocol): + """What the per-cell conceptual model needs, whatever routes its output. + + The part of the contract the lumped and distributed paths share. Split out so that + :class:`LumpedModel` does not advertise a need for a flow network it never touches. + + Attributes: + parameters: The conceptual model's parameters. A 3D `(rows, cols, n)` array for a + distributed run; a flat sequence for a lumped one. + lumped_model: The conceptual model instance whose `simulate` is called per cell. + initial_cond: Initial state values `[sp, sm, uz, lz, wc]`. + q_init: Initial discharge in m3/s, or None to let the model choose. + snow: 1 to run the snow routine, 0 otherwise. + area: Catchment area in km2. + conversion_factor: Depth-to-discharge factor for the temporal resolution. + dt: Time-step factor used by the Muskingum routing. + results: Where the run writes its output. None before the first run. + """ + + parameters: np.ndarray | list + lumped_model: BaseConceptualModel + initial_cond: list + q_init: float | None + snow: int + area: float | int + conversion_factor: float + dt: float + results: SimulationResults | None + + +class DistributedModel(ConceptualModelInputs, Protocol): + """What a distributed run requires on top of the conceptual model's own inputs. + + Attributes: + meteo: The three driver cubes and the calendar they cover. + flow_network: The routing network and the grid it defines. + date_index: The model's own calendar, which the drivers are checked against. + routing_method: Canonicalised routing method. `SpatialRouting` compares this against + `"Muskingum"` exactly to decide whether a cell is routed or skipped. + bankfull_depth: Read only when `routing_method` is not `"Muskingum"`; None otherwise. + """ + + meteo: MeteoInputs + flow_network: FlowNetwork + date_index: pd.DatetimeIndex + routing_method: str + bankfull_depth: np.ndarray | None + + +class LumpedModelInputs(ConceptualModelInputs, Protocol): + """What a lumped run requires: one column per variable rather than a grid. + + Attributes: + data: `(time, 4)` array of precipitation, ET, temperature and the long-term average. + maxbas: Whether the parameter vector carries a MAXBAS value, which changes how the + routing function is called. + temporal_resolution: `"daily"` or `"hourly"`. + start: First step of the simulation period. + end: Last step of the simulation period. + Qsim: Where the routed hydrograph lands. + """ + + data: np.ndarray + maxbas: bool + temporal_resolution: str + start: dt.datetime + end: dt.datetime + Qsim: Any + + +class FloodModel(DistributedModel, Protocol): + """A distributed model that also carries the river geometry the flood model reads. + + Attributes: + river_width: `(rows, cols)` channel width. + river_roughness: `(rows, cols)` channel roughness. + flood_plain_roughness: `(rows, cols)` floodplain roughness. + """ + + river_width: np.ndarray + river_roughness: np.ndarray + flood_plain_roughness: np.ndarray + + diff --git a/src/hapi/rrm/distrrm.py b/src/hapi/rrm/distrrm.py index 3cb936fa..94ed9a4b 100644 --- a/src/hapi/rrm/distrrm.py +++ b/src/hapi/rrm/distrrm.py @@ -33,15 +33,19 @@ def __init__(self): pass @staticmethod - def run_lumped_model(Model): + def run_lumped_model(Model) -> SimulationResults: """Run lumped rainfall-runoff model for every grid cell. Executes the lumped conceptual model (e.g., HBV) independently for each non-NaN cell in the catchment grid and converts the resulting discharge from mm/time-step to m3/s. - After execution the following attributes are set on *Model*: - `state_variables`, `quz`, and `qlz`. + Builds `Model.results` and returns it, so a caller that needs the arrays does + not have to read them back off the model and re-narrow them from `| None`. + + Returns: + SimulationResults: The results object, carrying `state_variables`, `quz` and + `qlz`, with `routing` still `RoutingKind.UNROUTED`. Args: Model (Catchment): A catchment model object carrying the following @@ -112,6 +116,7 @@ def run_lumped_model(Model): # convert quz and qlz from mm/time step to m3/sec # Timef*3.6 results.quz = results.quz * factor results.qlz = results.qlz * factor + return results @staticmethod def SpatialRouting(Model): @@ -226,7 +231,9 @@ def SpatialRouting(Model): # add the routed upstream flows to the current Quz in the cell results.quz_routed[x, y, :] = results.quz[x, y, :] + q_uzi - results.qlz_translated[x, y, :] = results.qlz[x, y, :] + qlzi + results.qlz_translated[x, y, :] = ( + results.qlz[x, y, :] + qlzi + ) results.Qtot = results.qlz_translated + results.quz_routed # Muskingum accumulates downstream, so a cell of `Qtot` is the discharge at that # cell and the outlet-cell shortcut in `extract_discharge` is valid. diff --git a/src/hapi/run.py b/src/hapi/run.py index 07f07ef0..64ef9d6e 100644 --- a/src/hapi/run.py +++ b/src/hapi/run.py @@ -4,30 +4,45 @@ both components of the spatial representation of the hydrological process (conceptual model and spatial routing) to calculate the predicted runoff at known locations based on a given performance function. + +`Run` is a namespace of static entry points, not a class to instantiate. Each one takes the +model it should run, validates it, and hands it to :class:`~hapi.wrapper.Wrapper`. What each +entry point requires is stated by the protocols below rather than by naming a concrete class: +:class:`~hapi.catchment.Catchment` satisfies them structurally, so this module does not import +it at runtime and anything else carrying the same attributes runs too. """ from __future__ import annotations from collections.abc import Callable from pathlib import Path -from typing import Any, NoReturn +from typing import TYPE_CHECKING, Any, NoReturn import numpy as np import pandas as pd from loguru import logger -from hapi.catchment import Catchment -from hapi.catchment import Lake as LakeType +from hapi.protocols import ( + DistributedModel, + FloodModel, + LumpedModelInputs, +) +from hapi.results import SimulationResults # from hapi.hm.saintvenant import SaintVenant from hapi.wrapper import Wrapper +if TYPE_CHECKING: + from hapi.catchment import Lake as LakeType + ROWS_MISMATCH_ERROR = "the parameters must have as many rows as the catchment grid" COLS_MISMATCH_ERROR = "the parameters must have as many columns as the catchment grid" -GRID_MISMATCH_ERROR = "all input data should have the same number of rows" +#: The flow-direction check tests both axes, so it says both. The entry points used to +#: carry three different wordings for this one check; the widest is the accurate one. +GRID_MISMATCH_ERROR = "all input data should have the same number of rows and columns" -def _check_parameters_cover_grid(model: Catchment) -> None: +def _check_parameters_cover_grid(model: DistributedModel) -> None: """Check the parameter array spans the catchment grid. The same two checks every distributed entry point makes before handing the model to the @@ -40,13 +55,14 @@ def _check_parameters_cover_grid(model: Catchment) -> None: Raises: ValueError: The parameter array has the wrong number of rows or columns. """ - if np.shape(model.parameters)[0] != model.flow_network.rows: + shape = np.asarray(model.parameters).shape + if shape[0] != model.flow_network.rows: raise ValueError(ROWS_MISMATCH_ERROR) - if np.shape(model.parameters)[1] != model.flow_network.cols: + if shape[1] != model.flow_network.cols: raise ValueError(COLS_MISMATCH_ERROR) -def _check_lake_meteo(model: Catchment, lake: LakeType) -> None: +def _check_lake_meteo(model: DistributedModel, lake: LakeType) -> None: """Check the lake's record lines up with the distributed drivers. Args: @@ -54,26 +70,68 @@ def _check_lake_meteo(model: Catchment, lake: LakeType) -> None: lake: The lake whose `MeteoData` is checked. Raises: - ValueError: The lake record is a different length from the distributed drivers, or - carries fewer than the three columns the lake model reads. + ValueError: The lake has no meteorological record, the record is a different length + from the distributed drivers, or it carries fewer than the three columns the + lake model reads. """ - if np.shape(lake.MeteoData)[0] != model.meteo.time_steps: + meteo_data = lake.MeteoData + if meteo_data is None: + raise ValueError( + "the lake has no meteorological data; call lake.read_meteo_data before " + "running a lake-aware entry point" + ) + if np.shape(meteo_data)[0] != model.meteo.time_steps: raise ValueError( "Lake meteorological data has to have the same length as the distributed " "raster data" ) - if np.shape(lake.MeteoData)[1] < 3: + if np.shape(meteo_data)[1] < 3: raise ValueError( "Lake Meteo data has to have at least three columns of rain, ET, and Temp" ) -class Run(Catchment): +def _validate_distributed(model: DistributedModel, check_flow_direction: bool) -> None: + """Run the checks every distributed entry point makes before the wrapper. + + Args: + model: The model about to run. + check_flow_direction: Whether to compare the flow-direction raster against the grid. + The MAXBAS paths never read that raster, so they do not require it. + + Raises: + ValueError: The grid, the drivers and the parameters do not agree. + """ + if check_flow_direction: + flow_dir_arr = model.flow_network.flow_dir_arr + # `FlowNetwork` takes the direction raster as optional because MAXBAS never reads + # it. The paths that route cell to cell do, so say which raster is missing rather + # than failing on None inside the routing loop. + if flow_dir_arr is None: + raise ValueError( + "this run routes cell to cell and needs a flow-direction raster, but the " + "flow network was built without one; pass it to FlowNetwork.from_rasters" + ) + fd_rows, fd_cols = flow_dir_arr.shape + if fd_rows != model.flow_network.rows or fd_cols != model.flow_network.cols: + raise ValueError(GRID_MISMATCH_ERROR) + + # The three cubes already agree with each other (checked when MeteoInputs was + # built); this is the other half -- that they cover the model's grid. + model.meteo.validate_against( + model.flow_network.rows, model.flow_network.cols, model.date_index + ) + _check_parameters_cover_grid(model) + + +class Run: """Run the catchment model. - The Run sub-class validates the spatial data and hands it to the - Wrapper class. It is a sub-class of the Catchment class, so you - need to create the Catchment object first to run the model. + A namespace of static entry points, not a class to instantiate. Each one validates the + model it is given and hands it to :class:`~hapi.wrapper.Wrapper`, returning the + :class:`~hapi.results.SimulationResults` the run produced. The same object is also + assigned to the model's `results`, so the result arrays stay readable off the model + afterwards. Methods: RunHapi: Run the distributed hydrological model. @@ -81,23 +139,39 @@ class Run(Catchment): runFW1: Run the FW1 distributed model. RunFW1withLake: Run the FW1 model with a lake component. runLumped: Run the lumped conceptual model. + RunFloodModel: Run the flood model. + + Examples: + - Build a model and run it; the results come back and stay on the model: + ```python + >>> from hapi.catchment import Catchment + >>> from hapi.routing import Routing + >>> from hapi.run import Run + >>> model = Catchment.from_yaml( + ... "examples/hydrological-model/coello/run/coello-lumped-model-run.yaml" + ... ) + >>> results = Run.runLumped(model, 1, Routing.muskingum_v) + >>> results.routing.value + 'lumped' + >>> results is model.results + True + + ``` + + See Also: + hapi.catchment.Catchment.from_yaml: Builds a model from a run configuration. """ - def __init__(self): - """Initialize the Run class.""" - self.Qsim: np.ndarray | pd.DataFrame | None = None - - @classmethod - def from_yaml(cls, path: str | Path) -> NoReturn: + @staticmethod + def from_yaml(path: str | Path) -> NoReturn: """Refuse to build a `Run`, explaining the pattern instead. - `Run` subclasses `Catchment` to hold its entry points, not to be a catchment: its - `__init__` takes no arguments, so the inherited `Catchment.from_yaml` could only fail - with a `TypeError` about constructor arity -- an error saying nothing about what to do - instead. The methods here are called on a model built elsewhere. + `Run` is a namespace of entry points, not a model: there is nothing for a + configuration to build. Kept as an explicit refusal because the message it gives is + more useful than the `AttributeError` that would replace it. Args: - path: Ignored; present so the signature matches the one it overrides. + path: Ignored; present so the call a caller is likely to try is answered. Raises: TypeError: Always. @@ -112,19 +186,6 @@ def from_yaml(cls, path: str | Path) -> NoReturn: ... print(str(error).split(";")[0]) Run cannot be built from a configuration - ``` - - Build the model with `Catchment.from_yaml` and hand it to the entry point: - ```python - >>> from hapi.catchment import Catchment - >>> from hapi.routing import Routing - >>> from hapi.run import Run - >>> model = Catchment.from_yaml( - ... "examples/hydrological-model/coello/run/coello-lumped-model-run.yaml" - ... ) - >>> Run.runLumped(model, 1, Routing.muskingum_v) - >>> len(model.Qsim) - 1095 - ``` See Also: @@ -136,7 +197,8 @@ def from_yaml(cls, path: str | Path) -> NoReturn: "it in, e.g. Run.RunHapi(model)." ) - def RunHapi(self): + @staticmethod + def RunHapi(model: DistributedModel) -> SimulationResults: """Run the distributed hydrological model. Validates that all input arrays (precipitation, evapotranspiration, @@ -144,93 +206,87 @@ def RunHapi(self): dimensions, then executes the rainfall-runoff model via the Wrapper. - The following instance attributes are set after execution: - - - `state_variables`: 4D array (rows, cols, time, states) where - states are [sp, wc, sm, uz, lv]. - - `qlz`: 3D array of the lower zone discharge. - - `quz`: 3D array of the upper zone discharge. - - `qout`: 1D timeseries of discharge at the catchment outlet - in m3/sec. - - `quz_routed`: 3D array of the upper zone discharge - accumulated and routed at each time step. - - `qlz_translated`: 3D array of the lower zone discharge - translated at each time step. + Args: + model: The model to run. See :class:`DistributedModel` for what it must carry. + + Returns: + SimulationResults: The run's output, also assigned to `model.results`: + + - `state_variables`: 4D array (rows, cols, time, states) where + states are [sp, wc, sm, uz, lv]. + - `qlz`: 3D array of the lower zone discharge. + - `quz`: 3D array of the upper zone discharge. + - `quz_routed`: 3D array of the upper zone discharge + accumulated and routed at each time step. + - `qlz_translated`: 3D array of the lower zone discharge + translated at each time step. + - `Qtot`: `quz_routed + qlz_translated`. Routed by Muskingum, so the outlet + cell carries the outlet hydrograph; `extract_discharge` fills `qout` from it. Raises: ValueError: If input data arrays have inconsistent row counts, column counts, or temporal lengths. """ - # input dimensions - fd_rows, fd_cols = self.flow_network.flow_dir_arr.shape - if fd_rows != self.flow_network.rows or fd_cols != self.flow_network.cols: - raise ValueError(GRID_MISMATCH_ERROR) - - # input dimensions - # The three cubes already agree with each other (checked when MeteoInputs was - # built); this is the other half -- that they cover the model's grid. - self.meteo.validate_against( - self.flow_network.rows, self.flow_network.cols, self.date_index - ) - _check_parameters_cover_grid(self) + _validate_distributed(model, check_flow_direction=True) # run the model - Wrapper.RRMModel(self) + results = Wrapper.RRMModel(model) - print("Model Run has finished") + logger.info("Model Run has finished") + return results - def RunFloodModel(self): + @staticmethod + def RunFloodModel(model: FloodModel) -> SimulationResults: """Run the flood model. Runs the conceptual distributed hydrological model with additional validation for river geometry inputs (bankfull depth, river width, river roughness, and flood plain roughness). + Args: + model: The model to run. See :class:`FloodModel` for what it must carry. + + Returns: + SimulationResults: The run's output, also assigned to `model.results`. + Raises: ValueError: If meteorological input arrays, parameter arrays, or river geometry arrays have inconsistent dimensions. """ - # input dimensions - [fd_rows, fd_cols] = self.flow_network.flow_dir_arr.shape - if fd_rows != self.flow_network.rows or fd_cols != self.flow_network.cols: - raise ValueError(GRID_MISMATCH_ERROR) - - # input dimensions - # The three cubes already agree with each other (checked when MeteoInputs was - # built); this is the other half -- that they cover the model's grid. - self.meteo.validate_against( - self.flow_network.rows, self.flow_network.cols, self.date_index - ) - _check_parameters_cover_grid(self) - if any( - np.shape(arr)[0] != self.flow_network.rows - for arr in ( - self.bankfull_depth, - self.river_width, - self.river_roughness, - self.flood_plain_roughness, + _validate_distributed(model, check_flow_direction=True) + + named_geometry = { + "bankfull_depth": model.bankfull_depth, + "river_width": model.river_width, + "river_roughness": model.river_roughness, + "flood_plain_roughness": model.flood_plain_roughness, + } + # `read_river_geometry` sets all four together, so a missing one means it was never + # called. Naming them beats `np.shape(None)` raising from inside the comparison. + missing = [name for name, arr in named_geometry.items() if arr is None] + if missing: + raise ValueError( + f"the flood model needs the river geometry, but {', '.join(missing)} " + "is not set; call read_river_geometry first" ) - ): + # Rebuilt from the non-None values rather than `.values()` directly: the guard above + # has already ruled None out, but only a comprehension carries that into the type. + geometry = [arr for arr in named_geometry.values() if arr is not None] + if any(np.shape(arr)[0] != model.flow_network.rows for arr in geometry): raise ValueError(GRID_MISMATCH_ERROR) - if any( - np.shape(arr)[1] != self.flow_network.cols - for arr in ( - self.bankfull_depth, - self.river_width, - self.river_roughness, - self.flood_plain_roughness, - ) - ): + if any(np.shape(arr)[1] != model.flow_network.cols for arr in geometry): raise ValueError("all input data should have the same number of columns") # run the model - Wrapper.RRMModel(self) - print("RRM has finished") + results = Wrapper.RRMModel(model) + logger.info("RRM has finished") # SV = SaintVenant() - # SV.KinematicRaster(self) + # SV.KinematicRaster(model) # print("1D model Run has finished") + return results - def runHAPIwithLake(self, lake: LakeType): + @staticmethod + def runHAPIwithLake(model: DistributedModel, lake: LakeType) -> SimulationResults: """Run the distributed model with a lake component. Validates that all input arrays have consistent dimensions and @@ -239,72 +295,67 @@ def runHAPIwithLake(self, lake: LakeType): the Wrapper. Args: + model: The model to run. See :class:`DistributedModel` for what it must carry. lake: Lake object containing lake configuration and meteorological data. Must have a `MeteoData` attribute with shape `(time_steps, >= 3)` where columns are rain, ET, and temperature. + Returns: + SimulationResults: The run's output, also assigned to `model.results`. + Raises: ValueError: If input data arrays have inconsistent dimensions or if the lake meteorological data length does not match the distributed raster data length. """ - # input dimensions - [fd_rows, fd_cols] = self.flow_network.flow_dir_arr.shape - if fd_rows != self.flow_network.rows or fd_cols != self.flow_network.cols: - raise ValueError( - "all input data should have the same number of rows and columns" - ) - - # input dimensions - # The three cubes already agree with each other (checked when MeteoInputs was - # built); this is the other half -- that they cover the model's grid. - self.meteo.validate_against( - self.flow_network.rows, self.flow_network.cols, self.date_index - ) - _check_parameters_cover_grid(self) - _check_lake_meteo(self, lake) + _validate_distributed(model, check_flow_direction=True) + _check_lake_meteo(model, lake) # run the model - Wrapper.RRMWithlake(self, lake) + results = Wrapper.RRMWithlake(model, lake) - print("Model Run has finished") + logger.info("Model Run has finished") + return results - def runFW1(self): + @staticmethod + def runFW1(model: DistributedModel) -> SimulationResults: """Run the FW1 distributed hydrological model. Validates that all input arrays have consistent dimensions, - then executes the FW1 model via the Wrapper. - - The following instance attributes are set after execution: - - - `st`: 4D array of state variables. - - `q_out`: 1D array of calculated discharge at the catchment - outlet, summed over every cell. - - `q_uz`: 3D array of distributed discharge for each cell. - - `Qtot`, `quz_routed`, `qlz_translated`: 3D per-cell fields - read by `save_results` and `plot_distributed_results`. MAXBAS - routes each cell straight to the outlet, so a cell of `Qtot` is - that cell's *contribution* to the outlet — `np.nansum` over the - domain reproduces `q_out`. Use - `extract_discharge(frame_work_1=True)`; the default outlet-cell - shortcut is invalid for this path and raises. + then executes the FW1 model via the Wrapper. The flow-direction + raster is not checked here because MAXBAS never reads it. + + Args: + model: The model to run. See :class:`DistributedModel` for what it must carry. + + Returns: + SimulationResults: The run's output, also assigned to `model.results`: + + - `state_variables`: 4D array of state variables. + - `qout`: 1D array of calculated discharge at the catchment + outlet, summed over every cell. + - `quz`: 3D array of distributed discharge for each cell. + - `Qtot`, `quz_routed`, `qlz_translated`: 3D per-cell fields + read by `save_results` and `plot_distributed_results`. MAXBAS + routes each cell straight to the outlet, so a cell of `Qtot` is + that cell's *contribution* to the outlet — `np.nansum` over the + domain reproduces `qout`. Use + `extract_discharge(frame_work_1=True)`; the default outlet-cell + shortcut is invalid for this path and raises. Raises: ValueError: If input data arrays have inconsistent row counts, column counts, or temporal lengths. """ - # The three cubes already agree with each other (checked when MeteoInputs was - # built); this is the other half -- that they cover the model's grid. - self.meteo.validate_against( - self.flow_network.rows, self.flow_network.cols, self.date_index - ) - _check_parameters_cover_grid(self) + _validate_distributed(model, check_flow_direction=False) # run the model - Wrapper.FW1(self) + results = Wrapper.FW1(model) - print("Model Run has finished") + logger.info("Model Run has finished") + return results - def RunFW1withLake(self, lake: LakeType): + @staticmethod + def RunFW1withLake(model: DistributedModel, lake: LakeType) -> SimulationResults: """Run the FW1 distributed model with a lake component. Validates that all input arrays have consistent dimensions and @@ -312,98 +363,69 @@ def RunFW1withLake(self, lake: LakeType): then executes the FW1 model with lake routing via the Wrapper. Args: + model: The model to run. See :class:`DistributedModel` for what it must carry. lake: Lake object containing lake configuration and meteorological data. Must have a `MeteoData` attribute with shape `(time_steps, >= 3)` where columns are rain, ET, and temperature. - Note: - The following catchment attributes should be set before - calling this method: - - - `prec_path`: Path to the folder containing precipitation - rasters. - - `evap_path`: Path to the folder containing - evapotranspiration rasters. - - `temp_path`: Path to the folder containing temperature - rasters. - - `flow_acc_path`: Path to the flow accumulation raster. - - `flow_direction_path`: Path to the flow direction raster. - - `ParPath`: Path to the folder containing parameter - rasters. - - `p2`: List of unoptimized parameters where `p2[0]` - is tfac and `p2[1]` is catchment area in km2. + Returns: + SimulationResults: The run's output, also assigned to `model.results`. Raises: ValueError: If input data arrays have inconsistent dimensions or if the lake meteorological data length does not match the distributed raster data length. """ - # input data validation - - # input dimensions - # The three cubes already agree with each other (checked when MeteoInputs was - # built); this is the other half -- that they cover the model's grid. - self.meteo.validate_against( - self.flow_network.rows, self.flow_network.cols, self.date_index - ) - _check_parameters_cover_grid(self) - _check_lake_meteo(self, lake) + _validate_distributed(model, check_flow_direction=False) + _check_lake_meteo(model, lake) # run the model - Wrapper.FW1Withlake(self, lake) + return Wrapper.FW1Withlake(model, lake) + @staticmethod def runLumped( - self, + model: LumpedModelInputs, Route: int = 0, routing_fn: Callable[..., Any] | None = None, - ): + ) -> SimulationResults: """Run the lumped conceptual model. Executes a lumped conceptual hydrological model, optionally routing the generated discharge hydrograph. The simulated - discharge is stored in `self.Qsim` as a pandas DataFrame + discharge is stored in `model.Qsim` as a pandas DataFrame indexed by the simulation date range. Args: + model: The model to run. See :class:`LumpedModelInputs` for what it must carry. Route: Flag to decide whether to route the generated discharge hydrograph. Use 0 for no routing or 1 to enable routing. Defaults to 0. routing_fn: Function to route the discharge hydrograph. - If None, an empty list is used. Defaults to None. - - Note: - The following attributes should be defined before calling - this method: - - - `LumpedModel`: Conceptual model containing a - `simulate` method. - - `data`: Numpy array of meteorological data with - columns for precipitation, evapotranspiration, - temperature, and long-term average temperature. - - `Parameters`: Numpy array of conceptual model - parameters. - - `CatArea`: Catchment area in km2. - - `conversion_factor`: Time conversion factor - (e.g., 24 for daily). - - `InitialCond`: List of initial state variable - values [sp, sm, uz, lz, wc]. - - `Snow`: Whether to use the snow subroutine (0 or 1). - - `q_init`: Initial discharge value. + Required when `Route` is not 0. + + Returns: + SimulationResults: The run's output, also assigned to `model.results`. A lumped + run applies no spatial routing, so the routed fields stay None and + `routing` is `RoutingKind.LUMPED`. + + Raises: + ValueError: `Route` is not 0 and no routing function was given. """ if routing_fn is None and Route != 0: raise ValueError("routing_fn must be a callable when Route != 0") - if self.temporal_resolution.lower() == "daily": - ind = pd.date_range(self.start, self.end, freq="D") + if model.temporal_resolution.lower() == "daily": + ind = pd.date_range(model.start, model.end, freq="D") else: - ind = pd.date_range(self.start, self.end, freq="h") + ind = pd.date_range(model.start, model.end, freq="h") Qsim = pd.DataFrame(index=ind) - Wrapper.Lumped(self, Route, routing_fn) - Qsim["q"] = self.Qsim - self.Qsim = Qsim[:] + results = Wrapper.Lumped(model, Route, routing_fn) + Qsim["q"] = model.Qsim + model.Qsim = Qsim[:] logger.info("Lumped model run has finished successfully") + return results if __name__ == "__main__": diff --git a/src/hapi/wrapper.py b/src/hapi/wrapper.py index a24920d3..0f0a8528 100644 --- a/src/hapi/wrapper.py +++ b/src/hapi/wrapper.py @@ -13,13 +13,14 @@ import numpy as np +from hapi.protocols import ConceptualModelInputs, DistributedModel, LumpedModelInputs from hapi.results import RoutingKind, SimulationResults from hapi.routing import Routing as routing from hapi.rrm.distrrm import DistributedRRM as distrrm from hapi.rrm.hbv_lake import HBVLake if TYPE_CHECKING: - from hapi.catchment import Catchment, Lake + from hapi.catchment import Lake class Wrapper: @@ -44,7 +45,7 @@ def __init__(self): pass @staticmethod - def RRMModel(Model: Catchment, ll_temp=None, q_0=None): + def RRMModel(Model: DistributedModel, ll_temp=None, q_0=None) -> SimulationResults: """Run the distributed rainfall-runoff model with spatial routing. Connects two modules: @@ -80,15 +81,18 @@ def RRMModel(Model: Catchment, ll_temp=None, q_0=None): Defaults to None. """ # run the rainfall runoff model separately - distrrm.run_lumped_model(Model) + results = distrrm.run_lumped_model(Model) # run the GIS part to rout from cell to another. It records # `RoutingKind.MUSKINGUM` on the results, which is what makes the outlet-cell # shortcut in `extract_discharge` valid for them. distrrm.SpatialRouting(Model) + return results @staticmethod - def RRMWithlake(Model: Catchment, Lake: Lake, ll_temp=None, q_0=None): + def RRMWithlake( + Model: DistributedModel, Lake: Lake, ll_temp=None, q_0=None + ) -> SimulationResults: """Run the distributed RRM with lake simulation and routing. Connects three modules: the lake module, the distributed @@ -148,7 +152,7 @@ def RRMWithlake(Model: Catchment, Lake: Lake, ll_temp=None, q_0=None): ) # subcatchment - distrrm.run_lumped_model(Model) + results = distrrm.run_lumped_model(Model) # routing lake discharge with DS cell k & x and adding to cell Q qlake = routing.muskingum_v( @@ -165,7 +169,7 @@ def RRMWithlake(Model: Catchment, Lake: Lake, ll_temp=None, q_0=None): # step here made it one longer than the array it is added to, which raised for every # input and left this entry point unrunnable. # both lake & Quz are in m3/s - quz = Model.results.quz + quz = results.quz quz[Lake.OutflowCell[0], Lake.OutflowCell[1], :] = ( quz[Lake.OutflowCell[0], Lake.OutflowCell[1], :] + qlake ) @@ -173,9 +177,10 @@ def RRMWithlake(Model: Catchment, Lake: Lake, ll_temp=None, q_0=None): # run the GIS part to rout from cell to another. It records # `RoutingKind.MUSKINGUM` on the results. distrrm.SpatialRouting(Model) + return results @staticmethod - def _set_maxbas_output_fields(Model: Catchment): + def _set_maxbas_output_fields(Model: ConceptualModelInputs) -> None: """Fill the distributed output fields after a triangular (MAXBAS) run. `save_results` and `plot_distributed_results` read `Qtot`, @@ -210,7 +215,7 @@ def _set_maxbas_output_fields(Model: Catchment): results.routing = RoutingKind.MAXBAS @staticmethod - def FW1(Model: Catchment, ll_temp=None, q_0=None): + def FW1(Model: DistributedModel, ll_temp=None, q_0=None) -> SimulationResults: """Run the distributed RRM with triangular function-1 routing. Connects two modules: @@ -235,13 +240,12 @@ def FW1(Model: Catchment, ll_temp=None, q_0=None): Defaults to None. """ # subcatchment - distrrm.run_lumped_model(Model) + results = distrrm.run_lumped_model(Model) distrrm.DistMaxbas1(Model) Wrapper._set_maxbas_output_fields(Model) - results = Model.results steps = Model.meteo.simulation_steps qlz1 = np.array( [np.nansum(results.qlz[:, :, i]) for i in range(steps)] @@ -251,9 +255,12 @@ def FW1(Model: Catchment, ll_temp=None, q_0=None): ) # average of all cells (routed mm/timestep) results.qout = (qlz1 + quz1)[:-1] + return results @staticmethod - def FW1Withlake(Model: Catchment, Lake: Lake, ll_temp=None, q_0=None): + def FW1Withlake( + Model: DistributedModel, Lake: Lake, ll_temp=None, q_0=None + ) -> SimulationResults: """Run the distributed RRM with lake and triangular routing. Connects three modules: @@ -316,7 +323,7 @@ def FW1Withlake(Model: Catchment, Lake: Lake, ll_temp=None, q_0=None): ) # subcatchment - distrrm.run_lumped_model(Model) + results = distrrm.run_lumped_model(Model) distrrm.DistMaxbas1(Model) @@ -324,7 +331,6 @@ def FW1Withlake(Model: Catchment, Lake: Lake, ll_temp=None, q_0=None): # extent, so it enters `qout` below but never `Qtot`. Wrapper._set_maxbas_output_fields(Model) - results = Model.results steps = Model.meteo.simulation_steps qlz1 = np.array( [np.nansum(results.qlz[:, :, i]) for i in range(steps)] @@ -341,9 +347,12 @@ def FW1Withlake(Model: Catchment, Lake: Lake, ll_temp=None, q_0=None): # `qout[:-1]` -- dropping the trailing slot, not the leading initial-state one. The # lake series has to be trimmed the same way or the two cannot be added at all. results.qout = qout[:-1] + Lake.QlakeR[:-1] + return results @staticmethod - def Lumped(Model: Catchment, Routing: int = 0, RoutingFn: Callable | None = None): + def Lumped( + Model: LumpedModelInputs, Routing: int = 0, RoutingFn: Callable | None = None + ) -> SimulationResults: """Run a lumped conceptual model with optional routing. Executes a lumped rainfall-runoff model (e.g., HBV) to @@ -418,14 +427,15 @@ def Lumped(Model: Catchment, Routing: int = 0, RoutingFn: Callable | None = None factor = Model.area / Model.conversion_factor # A lumped run has no spatial routing at all, so the routed fields stay None and # the routing kind says why -- rather than a MAXBAS flag left over from elsewhere. - Model.results = SimulationResults( + results = SimulationResults( routing=RoutingKind.LUMPED, quz=quz * factor, qlz=qlz * factor, state_variables=state_variables, ) + Model.results = results - Model.Qsim = Model.results.quz + Model.results.qlz + Model.Qsim = results.quz + results.qlz if Routing != 0 and Model.maxbas: Model.Qsim = RoutingFn(np.array(Model.Qsim[:-1]), Model.parameters[-1]) @@ -437,6 +447,7 @@ def Lumped(Model: Catchment, Routing: int = 0, RoutingFn: Callable | None = None Model.parameters[-1], Model.dt, ) + return results if __name__ == "__main__": diff --git a/tests/rrm/catchment/test_run_results_coupling.py b/tests/rrm/catchment/test_run_results_coupling.py new file mode 100644 index 00000000..e9b7510e --- /dev/null +++ b/tests/rrm/catchment/test_run_results_coupling.py @@ -0,0 +1,380 @@ +"""Tests for how `Run` and `Catchment` are coupled, and for the results object between them. + +`Run` used to subclass `Catchment` and be called unbound (`Run.RunHapi(model)`, with the +catchment landing on `self`), writing nine result arrays back onto the model plus a private +`_maxbas_routed` flag recording which routing had produced them. It is now a namespace of +static entry points that state what they need as a protocol and return a +`SimulationResults`. + +These tests pin the three properties that change buys: the entry points do not depend on +`Catchment`, the results are one object rather than nine attributes, and the routing +provenance travels with the arrays instead of as a flag that the next run has to clear. +""" + +from __future__ import annotations + +import inspect +import subprocess +import sys + +import numpy as np +import pytest + +from hapi.catchment import Catchment +from hapi.inputs import FlowNetwork, MeteoInputs +from hapi.results import RoutingKind, SimulationResults +from hapi.rrm.hbv_bergestrom92 import HBVBergestrom92 as HBVLumped +from hapi.run import Run + +DATE_REGEX = r"\d{4}.\d{2}.\d{2}" + +RESULT_FIELDS = ( + "quz", + "qlz", + "state_variables", + "quz_routed", + "qlz_translated", + "Qtot", + "qout", +) + + +def _build(name: str, parameters: str, **fixtures) -> Catchment: + """Assemble a distributed Coello catchment ready to run. + + Args: + name: Catchment name. + parameters: Path to the parameter folder, which decides maxbas vs muskingum. + **fixtures: The `coello_*` fixture values the build reads. + + Returns: + Catchment: A model with meteo, flow network, parameters and conceptual model set. + """ + model = Catchment( + name, + fixtures["start"], + fixtures["end"], + spatial_resolution="Distributed", + temporal_resolution="Daily", + ) + model.meteo = MeteoInputs.from_rasters( + fixtures["prec"], + fixtures["temp"], + fixtures["evap"], + start=fixtures["start"], + end=fixtures["end"], + regex_string=DATE_REGEX, + date=True, + file_name_data_fmt="%Y.%m.%d", + ) + model.flow_network = FlowNetwork.from_rasters(fixtures["acc"], fixtures.get("fd")) + model.read_parameters(parameters, False, maxbas=fixtures.get("maxbas", False)) + model.read_lumped_model(HBVLumped, fixtures["area"], fixtures["initial_cond"]) + return model + + +@pytest.fixture +def coello_fixtures( + coello_start_date: str, + coello_end_date: str, + coello_prec_path: str, + coello_temp_path: str, + coello_evap_path: str, + coello_acc_path: str, + coello_fd_path: str, + coello_cat_area: int, + coello_initial_cond: list, +) -> dict: + """Bundle the `coello_*` fixtures the builder reads, so each test names one thing.""" + return { + "start": coello_start_date, + "end": coello_end_date, + "prec": coello_prec_path, + "temp": coello_temp_path, + "evap": coello_evap_path, + "acc": coello_acc_path, + "fd": coello_fd_path, + "area": coello_cat_area, + "initial_cond": coello_initial_cond, + } + + +class TestRunIsNotACatchment: + """The coupling itself: `Run` no longer inherits from or imports `Catchment`.""" + + def test_run_does_not_subclass_catchment(self): + """Test that `Run` is a plain namespace rather than a `Catchment` subclass. + + Test scenario: + The inheritance existed only so a catchment could be passed as `self`. Nothing + ever instantiated `Run` or checked `isinstance(x, Run)`, so the IS-A was never + true; asserting it is gone stops it coming back. + """ + assert not issubclass(Run, Catchment), ( + "Run must not subclass Catchment; it takes the model as an explicit parameter" + ) + assert Catchment not in Run.__mro__, ( + f"Catchment must not be in Run's MRO, got {Run.__mro__}" + ) + + @pytest.mark.parametrize( + "name", + [ + "RunHapi", + "RunFloodModel", + "runHAPIwithLake", + "runFW1", + "RunFW1withLake", + "runLumped", + "from_yaml", + ], + ) + def test_every_entry_point_is_a_static_method(self, name: str): + """Test that each entry point is a staticmethod taking the model explicitly. + + Test scenario: + An instance method would reintroduce the unbound-call pattern, where the first + parameter is named `self` but is really the model. A staticmethod cannot. + + Args: + name: The entry point being checked. + """ + attribute = inspect.getattr_static(Run, name) + assert isinstance(attribute, staticmethod), ( + f"Run.{name} must be a staticmethod, got {type(attribute).__name__}" + ) + first = next(iter(inspect.signature(getattr(Run, name)).parameters)) + assert first != "self", ( + f"Run.{name}'s first parameter must not be named self, got {first!r}" + ) + + def test_run_does_not_import_catchment_at_runtime(self): + """Test that importing `hapi.run` does not pull in `hapi.catchment`. + + Test scenario: + The dependency is inverted: `Run` owns protocols describing what it needs, and + `Catchment` satisfies them structurally. A runtime import would mean the + inversion is only cosmetic. Checked in a subprocess rather than by clearing + `sys.modules` here, which would leave every later test importing a second copy + of the package. + """ + probe = ( + "import sys; import hapi.run; " + "print('catchment-imported' if 'hapi.catchment' in sys.modules else 'clean')" + ) + completed = subprocess.run( + [sys.executable, "-c", probe], + capture_output=True, + text=True, + check=True, + ) + + assert completed.stdout.strip() == "clean", ( + "importing hapi.run must not import hapi.catchment; the protocols in " + "hapi.protocols exist so the run layer does not depend on the concrete class" + ) + + def test_from_yaml_refuses_and_names_the_pattern(self): + """Test that `Run.from_yaml` explains itself rather than raising AttributeError. + + Test scenario: + `Run` no longer inherits `Catchment.from_yaml`, so the call would fail with a + bare AttributeError. The explicit refusal is kept because it names what to do. + """ + with pytest.raises(TypeError, match="cannot be built from a configuration"): + Run.from_yaml("anything.yaml") + + +class TestEntryPointsReturnTheirResults: + """Entry points return a `SimulationResults` rather than only mutating the model.""" + + def test_run_hapi_returns_the_object_it_put_on_the_model( + self, coello_fixtures: dict, coello_dist_parameters_muskingum: str + ): + """Test that the returned results are the same object assigned to `model.results`. + + Test scenario: + Returning the results is what lets a caller work without reaching back into the + model. It must be the same object, not a copy, so the historical attribute reads + and the returned value can never disagree. + """ + model = _build("coello", coello_dist_parameters_muskingum, **coello_fixtures) + + results = Run.RunHapi(model) + + assert isinstance(results, SimulationResults), ( + f"RunHapi must return SimulationResults, got {type(results).__name__}" + ) + assert results is model.results, ( + "the returned results must be the same object assigned to model.results" + ) + assert results.routing is RoutingKind.MUSKINGUM, ( + f"a RunHapi run is Muskingum-routed, got {results.routing}" + ) + + def test_run_fw1_returns_maxbas_routed_results( + self, coello_fixtures: dict, coello_dist_parameters_maxbas: str + ): + """Test that the triangular path records MAXBAS on the results it returns. + + Test scenario: + The routing kind is what tells `extract_discharge` whether the outlet-cell + shortcut is valid, so the FW1 path must record it on the arrays it produced. + """ + model = _build( + "coello", coello_dist_parameters_maxbas, maxbas=True, **coello_fixtures + ) + + results = Run.runFW1(model) + + assert results.routing is RoutingKind.MAXBAS, ( + f"a runFW1 run is MAXBAS-routed, got {results.routing}" + ) + assert not results.outlet_shortcut_valid, ( + "MAXBAS sends every cell to the outlet, so the outlet-cell shortcut is invalid" + ) + + +class TestResultAttributesAreReadOnlyViews: + """The historical attribute names still read, but the results object owns the arrays.""" + + @pytest.mark.parametrize("field", RESULT_FIELDS) + def test_field_reads_through_to_the_results_object( + self, coello_fixtures: dict, coello_dist_parameters_muskingum: str, field: str + ): + """Test that each historical name returns exactly what the results object holds. + + Test scenario: + Existing scripts and notebooks read `model.Qtot` and friends after a run. The + properties exist so that keeps working; identity is asserted rather than + equality so a copy cannot pass. + + Args: + field: The result attribute being checked. + """ + model = _build("coello", coello_dist_parameters_muskingum, **coello_fixtures) + Run.RunHapi(model) + + assert getattr(model, field) is getattr(model.results, field), ( + f"model.{field} must read through to results.{field}, not copy it" + ) + + @pytest.mark.parametrize("field", RESULT_FIELDS) + def test_field_is_none_before_a_run(self, field: str): + """Test that the result names read as None on a catchment that has not run. + + Test scenario: + They used to be None-initialised attributes. Reading one before a run must stay + a None rather than becoming an AttributeError. + + Args: + field: The result attribute being checked. + """ + model = Catchment("empty", "2009-01-01", "2009-01-10") + + assert getattr(model, field) is None, ( + f"model.{field} must be None before a run, got {type(getattr(model, field))}" + ) + + @pytest.mark.parametrize("field", RESULT_FIELDS) + def test_field_cannot_be_assigned(self, field: str): + """Test that a result field rejects assignment. + + Test scenario: + These are outputs. A run that can be half-overwritten by hand is exactly what + the results object exists to prevent, so the properties have no setter and + staging a state goes through `model.results` instead. + + Args: + field: The result attribute being checked. + """ + model = Catchment("empty", "2009-01-01", "2009-01-10") + + with pytest.raises(AttributeError): + setattr(model, field, np.zeros((2, 2, 2))) + + +class TestRoutingProvenanceReplacesTheFlag: + """`_maxbas_routed` is derived from the results, so it cannot outlive the run.""" + + def test_a_muskingum_run_after_a_maxbas_run_clears_the_maxbas_reading( + self, + coello_fixtures: dict, + coello_dist_parameters_maxbas: str, + coello_dist_parameters_muskingum: str, + ): + """Test that running MAXBAS then Muskingum leaves the model reading as Muskingum. + + Test scenario: + This is the case the old boolean needed hand-clearing for: `_maxbas_routed` was + set by the MAXBAS path and had to be reset by every Muskingum path, with a + comment saying so. Deriving it from the results makes that impossible to forget, + because a new run replaces the object the reading comes from. + """ + model = _build( + "coello", coello_dist_parameters_maxbas, maxbas=True, **coello_fixtures + ) + Run.runFW1(model) + assert model._maxbas_routed, "the FW1 run should read as MAXBAS-routed" + + # Re-read the parameters the Muskingum path needs, then run it on the same model. + model.read_parameters(coello_dist_parameters_muskingum, False) + Run.RunHapi(model) + + assert not model._maxbas_routed, ( + "after a Muskingum run the model must no longer read as MAXBAS-routed" + ) + assert model.results.outlet_shortcut_valid, ( + "the outlet-cell shortcut is valid again once Muskingum has routed the results" + ) + + def test_a_fresh_catchment_does_not_read_as_maxbas_routed(self): + """Test that a model that has never run does not claim MAXBAS routing. + + Test scenario: + The derived reading has to answer for the no-results case too, since + `extract_discharge` consults it before checking anything else. + """ + model = Catchment("empty", "2009-01-01", "2009-01-10") + + assert not model._maxbas_routed, ( + "a catchment with no results must not read as MAXBAS-routed" + ) + + +class TestValidationNamesWhatIsMissing: + """The guards the protocols exposed: an optional input that this path does require.""" + + def test_a_muskingum_run_without_a_flow_direction_raster_says_so( + self, coello_fixtures: dict, coello_dist_parameters_muskingum: str + ): + """Test that cell-to-cell routing without a flow-direction raster raises clearly. + + Test scenario: + `FlowNetwork` takes the direction raster as optional because MAXBAS never reads + it, so a Muskingum run can be assembled without one. It used to fail on None + inside the validation; it now names the missing raster. + """ + fixtures = dict(coello_fixtures, fd=None) + model = _build("coello", coello_dist_parameters_muskingum, **fixtures) + + with pytest.raises(ValueError, match="flow-direction raster"): + Run.RunHapi(model) + + def test_the_flood_model_names_the_river_geometry_it_lacks( + self, coello_fixtures: dict, coello_dist_parameters_muskingum: str + ): + """Test that the flood model reports which geometry rasters are unset. + + Test scenario: + `read_river_geometry` sets four arrays together, so a missing one means it was + never called. The check used to reach `np.shape(None)`; it now lists the names. + """ + model = _build("coello", coello_dist_parameters_muskingum, **coello_fixtures) + + with pytest.raises(ValueError, match="read_river_geometry") as exc_info: + Run.RunFloodModel(model) + + assert "bankfull_depth" in str(exc_info.value), ( + f"the error should name the missing rasters, got: {exc_info.value}" + ) From 65935561f04932ca244fc35461da4f887dfdf235 Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Tue, 1 Sep 2026 23:09:17 +0200 Subject: [PATCH 05/54] refactor!: drop every compatibility shim and rename the legacy API The previous two commits kept old call sites working on purpose: result attributes stayed readable off the catchment through forwarding properties, and the CamelCase entry points were left alone because renaming them would break downstream users. Neither concession is wanted -- this is a redesign, so where the new design says something different, the old thing goes. Results have one home. The eight forwarding properties on `Catchment` (`quz`, `qlz`, `state_variables`, `quz_routed`, `qlz_translated`, `q_total`, `qout`) and the derived `_maxbas_routed` are deleted; the arrays are read as `model.results.q_total`. `Qtot` becomes `q_total`, since new code follows the naming convention like everything else. `Run.from_yaml` is deleted too: it existed to intercept a call the old inheritance made resolvable, and with the inheritance gone the absent attribute is the honest answer. Every legacy CamelCase entry point is renamed: Run.RunHapi -> Run.run_distributed Run.runFW1 -> Run.run_maxbas Run.runHAPIwithLake -> Run.run_distributed_with_lake Run.RunFW1withLake -> Run.run_maxbas_with_lake Run.RunFloodModel -> Run.run_flood Run.runLumped -> Run.run_lumped Wrapper.RRMModel -> Wrapper.run_muskingum Wrapper.RRMWithlake -> Wrapper.run_muskingum_with_lake Wrapper.FW1 -> Wrapper.run_maxbas Wrapper.FW1Withlake -> Wrapper.run_maxbas_with_lake Wrapper.Lumped -> Wrapper.run_lumped DistributedRRM.SpatialRouting -> route_muskingum DistributedRRM.DistMaxbas1 -> route_maxbas DistributedRRM.DistMaxbas2 -> route_maxbas_by_path_length Calibration.FW1Calibration -> Calibration.calibrate_maxbas Calibration.lumpedCalibration -> Calibration.calibrate_lumped `DistributedRRM.Dist_HBV2` is deleted rather than renamed: it is a self-contained legacy reimplementation of run_lumped_model plus route_muskingum that nothing calls. `extract_discharge` loses `frame_work_1` and `only_outlet`. The first asked the caller to restate which entry point they had just called, and raised when they got it wrong; the routing is a property of the arrays, so it is read off `results.routing` instead and the right hydrograph is selected automatically. The second was documented as having no effect at all. Both are gone, and with them the ValueError that existed only to catch a mis-set flag. Examples, notebooks, docs and README are updated throughout. BREAKING CHANGE: no aliases and no deprecation period. Result arrays move from `model.` to `model.results.`, `Qtot` is `q_total`, every entry point listed above is renamed, `Run.from_yaml` and `DistributedRRM.Dist_HBV2` are removed, and `extract_discharge` no longer accepts `frame_work_1` or `only_outlet`. --- README.md | 4 +- docs/api/catchment.md | 4 +- docs/examples/distributed-model-calib.md | 4 +- docs/examples/lumped-model-calibration.md | 2 +- docs/examples/lumped-model-run.md | 4 +- docs/examples/run-configuration.md | 4 +- ...boa-distributed-model-muskingum-lake.ipynb | 2 +- .../Note books/Lumped-Model_Run.ipynb | 2 +- .../Note books/Lumped_Model_Calib.ipynb | 4 +- .../Note books/check-03Jiboa-colab.ipynb | 4 +- .../Note books/check-03Jiboa.ipynb | 4 +- .../Note books/check-colab/Coello.ipynb | 2 +- ...stributed-model-muskingum-lake-colab.ipynb | 4 +- .../Note books/check-colab/Jiboa.ipynb | 2 +- ...ello-distributed-model-run-muskingum.ipynb | 8 +- .../coello-lumped-model-run-muskingum.ipynb | 2 +- .../Note books/lumped-model-run-coello.ipynb | 2 +- ...libration-deap-multiobjective-NSE-NSEHF.py | 4 +- ...alibration-deap-multiobjective-NSE-RMSE.py | 4 +- .../coello-lumped-model-calibration-deap.py | 4 +- .../coello-lumped-model-calibration.py | 4 +- .../coello-distributed-model-run-maxbas.py | 6 +- .../coello-distributed-model-run-netcdf.py | 4 +- .../run/coello-lumped-model-run-maxbas.py | 4 +- .../coello/run/coello-lumped-model-run.py | 4 +- .../Jiboa-distributed-model-muskingum-lake.py | 4 +- src/hapi/calibration.py | 55 ++-- src/hapi/catchment.py | 186 ++++--------- src/hapi/config.py | 8 +- src/hapi/inputs.py | 2 +- src/hapi/protocols.py | 4 +- src/hapi/results.py | 14 +- src/hapi/rrm/distrrm.py | 258 +----------------- src/hapi/run.py | 90 ++---- src/hapi/wrapper.py | 44 +-- tests/calibration/lumped_calibration.py | 4 +- .../test_calibration_distributed.py | 42 +-- tests/rrm/calibration/test_rrm_calibration.py | 2 +- tests/rrm/catchment/test_config.py | 35 +-- .../catchment/test_e2e_coello_from_netcdf.py | 37 ++- .../test_extract_discharge_distributed.py | 8 +- tests/rrm/catchment/test_flow_network.py | 2 +- tests/rrm/catchment/test_fw1_output_fields.py | 80 +++--- .../catchment/test_maxbas_routing_variants.py | 56 ++-- tests/rrm/catchment/test_meteo_inputs.py | 16 +- tests/rrm/catchment/test_plot_animation.py | 8 +- .../test_read_parameters_validation.py | 2 +- tests/rrm/catchment/test_rrm_catchment.py | 50 ++-- .../catchment/test_run_results_coupling.py | 167 ++++++------ tests/rrm/catchment/test_run_validation.py | 60 ++-- .../test_save_results_distributed.py | 4 +- tests/rrm/catchment/test_wrapper_with_lake.py | 79 +++--- tests/run/distributed_mode_run.py | 2 +- tests/run/lumped_run.py | 2 +- tests/sensitivity_analysis.py | 8 +- 55 files changed, 553 insertions(+), 868 deletions(-) diff --git a/README.md b/README.md index 5f91516d..99e6542c 100644 --- a/README.md +++ b/README.md @@ -126,5 +126,5 @@ Quick start - class names: PascalCase (Model, MyClass). - class method/function: snake_case (get_file, read_config). They should have a verb in them, because they perform some action. -Some CamelCase entry points survive from earlier releases (for example `Run.RunHapi` and `Wrapper.RRMModel`) -because examples and downstream code still call them. New methods are written in snake_case. +The CamelCase entry points that survived from earlier releases (`Run.RunHapi`, `Wrapper.RRMModel` and the rest) +have been renamed to snake_case. There is no compatibility alias: the names above are the only ones. diff --git a/docs/api/catchment.md b/docs/api/catchment.md index 9ac34321..8a664bd7 100644 --- a/docs/api/catchment.md +++ b/docs/api/catchment.md @@ -9,13 +9,13 @@ and stored in the one spelling the internals compare against: |---|---|---| | `muskingum` | `Muskingum` | Cell to cell along the flow-direction network. | | `maxbas` | `MAXBAS` | Every cell straight to the outlet through a triangular function. | -| `kinematic` | `Kinematic` | The flood model's own path (`Run.RunFloodModel`). | +| `kinematic` | `Kinematic` | The flood model's own path (`Run.run_flood`). | Anything else raises a `ValueError` naming the three. Up to and including version 1.7.0 the constructor stored whatever string it was handed, so a run configured as `"Max_bas"` — or as a descriptive label such as `"Muskingum-Cunge"` — was accepted and then silently routed with Muskingum, because -`distrrm.SpatialRouting` compares against `"Muskingum"` exactly. Rejecting the spelling is what +`distrrm.route_muskingum` compares against `"Muskingum"` exactly. Rejecting the spelling is what makes that comparison trustworthy; a script passing a spelling outside the table has to be updated to one of the three. diff --git a/docs/examples/distributed-model-calib.md b/docs/examples/distributed-model-calib.md index cb749828..f2bdddf5 100644 --- a/docs/examples/distributed-model-calib.md +++ b/docs/examples/distributed-model-calib.md @@ -100,11 +100,11 @@ Coello.read_discharge_gauges(GaugesPath, column='id', fmt="%Y-%m-%d") - The `Run` object connects all the components of the simulation together, the `Catchment` object, the `Lake` object and the `distributedrouting` object -- import the Run object and use the `Catchment` object as a parameter to the `Run` object, then call the RunHapi method to start the simulation +- import the Run object and use the `Catchment` object as a parameter to the `Run` object, then call the run_distributed method to start the simulation ```python from hapi.run import Run -Run.RunHapi(Coello) +Run.run_distributed(Coello) ``` - the result of the simulation will be stored as attributes in the Catchment object as follow diff --git a/docs/examples/lumped-model-calibration.md b/docs/examples/lumped-model-calibration.md index e772cb0d..f3858b13 100644 --- a/docs/examples/lumped-model-calibration.md +++ b/docs/examples/lumped-model-calibration.md @@ -79,7 +79,7 @@ To calibrate the HBV lumped model inside Hapi you need to follow the same steps - Run Calibration ```python - cal_parameters = Coello.lumpedCalibration(Basic_inputs, OptimizationArgs, print_error=None) + cal_parameters = Coello.calibrate_lumped(Basic_inputs, OptimizationArgs, print_error=None) print("Objective Function = " + str(round(cal_parameters[0],2))) print("Parameters are " + str(cal_parameters[1])) diff --git a/docs/examples/lumped-model-run.md b/docs/examples/lumped-model-run.md index f3dcb24b..b5c9442f 100644 --- a/docs/examples/lumped-model-run.md +++ b/docs/examples/lumped-model-run.md @@ -64,10 +64,10 @@ Coello.read_parameters(Parameterpath, Snow) RoutingFn = Routing.muskingum_v Route = 1 ``` -- now all the data required for the model are prepared in the right form, now you can call the `runLumped` wrapper to initiate the calculation +- now all the data required for the model are prepared in the right form, now you can call the `run_lumped` wrapper to initiate the calculation ```python -Run.runLumped(Coello, Route, RoutingFn) +Run.run_lumped(Coello, Route, RoutingFn) ``` to calculate some metrics for the quality assessment of the calculate discharge the `statista.descriptors` contains some metrics like `rmse`, `nse`, `kge` and `wb` , you need to load it, a measured time series of doscharge for the same diff --git a/docs/examples/run-configuration.md b/docs/examples/run-configuration.md index 98180ecb..015ea830 100644 --- a/docs/examples/run-configuration.md +++ b/docs/examples/run-configuration.md @@ -14,7 +14,7 @@ from hapi.catchment import Catchment from hapi.run import Run Coello = Catchment.from_yaml("coello-lumped-model-run.yaml") -Run.runLumped(Coello, Routing.triangular_routing_1) +Run.run_lumped(Coello, Routing.triangular_routing_1) ``` The four shipped examples under `examples/hydrological-model/coello/run/` are each a pair — a @@ -158,7 +158,7 @@ Coello.save_results( The schema describes a `Catchment` run. It carries no field for a lake record, a river geometry, or a flow-path-length raster, so lake-aware runs (`Run.RunHapiwithLake`), the flood model -(`Run.RunFloodModel`) and `DistMaxbas2` are still assembled in Python. +(`Run.run_flood`) and `route_maxbas_by_path_length` are still assembled in Python. `Calibration.from_yaml` works — it takes the same constructor arguments — and gives back a `Calibration` to call the calibration methods on. `Run.from_yaml` does not: `Run` holds entry diff --git a/examples/hydrological-model/Note books/Jiboa-distributed-model-muskingum-lake.ipynb b/examples/hydrological-model/Note books/Jiboa-distributed-model-muskingum-lake.ipynb index 2d00c7ad..2461a6f1 100644 --- a/examples/hydrological-model/Note books/Jiboa-distributed-model-muskingum-lake.ipynb +++ b/examples/hydrological-model/Note books/Jiboa-distributed-model-muskingum-lake.ipynb @@ -242,7 +242,7 @@ "metadata": {}, "outputs": [], "source": [ - "Run.runHAPIwithLake(Jiboa, JiboaLake)" + "Run.run_distributed_with_lake(Jiboa, JiboaLake)" ] }, { diff --git a/examples/hydrological-model/Note books/Lumped-Model_Run.ipynb b/examples/hydrological-model/Note books/Lumped-Model_Run.ipynb index 87ba5bc3..c6a624f7 100644 --- a/examples/hydrological-model/Note books/Lumped-Model_Run.ipynb +++ b/examples/hydrological-model/Note books/Lumped-Model_Run.ipynb @@ -208,7 +208,7 @@ "metadata": {}, "outputs": [], "source": [ - "Run.runLumped(Coello, Route, RoutingFn)" + "Run.run_lumped(Coello, Route, RoutingFn)" ] }, { diff --git a/examples/hydrological-model/Note books/Lumped_Model_Calib.ipynb b/examples/hydrological-model/Note books/Lumped_Model_Calib.ipynb index 9aa02740..b7c7b924 100644 --- a/examples/hydrological-model/Note books/Lumped_Model_Calib.ipynb +++ b/examples/hydrological-model/Note books/Lumped_Model_Calib.ipynb @@ -261,7 +261,7 @@ "metadata": {}, "outputs": [], "source": [ - "cal_parameters = Coello.lumpedCalibration(\n", + "cal_parameters = Coello.calibrate_lumped(\n", " Basic_inputs, OptimizationArgs, printError=None\n", ")\n", "\n", @@ -296,7 +296,7 @@ "outputs": [], "source": [ "Coello.parameters = cal_parameters[1]\n", - "Run.runLumped(Coello, Route, RoutingFn)" + "Run.run_lumped(Coello, Route, RoutingFn)" ] }, { diff --git a/examples/hydrological-model/Note books/check-03Jiboa-colab.ipynb b/examples/hydrological-model/Note books/check-03Jiboa-colab.ipynb index 241121db..7154062d 100644 --- a/examples/hydrological-model/Note books/check-03Jiboa-colab.ipynb +++ b/examples/hydrological-model/Note books/check-03Jiboa-colab.ipynb @@ -127,7 +127,7 @@ "import pandas as pd\n", "\n", "# HAPI modules\n", - "from Hapi.run import runHAPIwithLake" + "from Hapi.run import run_distributed_with_lake" ] }, { @@ -258,7 +258,7 @@ "outputs": [], "source": [ "Sim = pd.DataFrame(index=lakeCalib.index)\n", - "st, Sim['Q'], q_uz_routed, q_lz_trans = runHAPIwithLake(\n", + "st, Sim['Q'], q_uz_routed, q_lz_trans = run_distributed_with_lake(\n", " HBV,\n", " Paths,\n", " ParPath,\n", diff --git a/examples/hydrological-model/Note books/check-03Jiboa.ipynb b/examples/hydrological-model/Note books/check-03Jiboa.ipynb index b922dc24..0c77a680 100644 --- a/examples/hydrological-model/Note books/check-03Jiboa.ipynb +++ b/examples/hydrological-model/Note books/check-03Jiboa.ipynb @@ -55,7 +55,7 @@ "import pandas as pd\n", "\n", "# HAPI modules\n", - "from Hapi.rrm import runHAPIwithLake\n", + "from Hapi.rrm import run_distributed_with_lake\n", "from osgeo import gdal" ] }, @@ -137,7 +137,7 @@ "outputs": [], "source": [ "Sim = pd.DataFrame(index=lakeCalib.index)\n", - "st, Sim['Q'], q_uz_routed, q_lz_trans = runHAPIwithLake(\n", + "st, Sim['Q'], q_uz_routed, q_lz_trans = run_distributed_with_lake(\n", " HBV,\n", " Paths,\n", " ParPath,\n", diff --git a/examples/hydrological-model/Note books/check-colab/Coello.ipynb b/examples/hydrological-model/Note books/check-colab/Coello.ipynb index 85f2a341..f516152d 100644 --- a/examples/hydrological-model/Note books/check-colab/Coello.ipynb +++ b/examples/hydrological-model/Note books/check-colab/Coello.ipynb @@ -233,7 +233,7 @@ }, "outputs": [], "source": [ - "Run.RunHapi(Coello)" + "Run.run_distributed(Coello)" ] }, { diff --git a/examples/hydrological-model/Note books/check-colab/Jiboa-distributed-model-muskingum-lake-colab.ipynb b/examples/hydrological-model/Note books/check-colab/Jiboa-distributed-model-muskingum-lake-colab.ipynb index 6852aeeb..c04ecd27 100644 --- a/examples/hydrological-model/Note books/check-colab/Jiboa-distributed-model-muskingum-lake-colab.ipynb +++ b/examples/hydrological-model/Note books/check-colab/Jiboa-distributed-model-muskingum-lake-colab.ipynb @@ -126,7 +126,7 @@ "import pandas as pd\n", "\n", "# HAPI modules\n", - "from Hapi.run import runHAPIwithLake\n", + "from Hapi.run import run_distributed_with_lake\n", "from osgeo import gdal" ] }, @@ -258,7 +258,7 @@ "outputs": [], "source": [ "Sim = pd.DataFrame(index=lakeCalib.index)\n", - "st, Sim['Q'], q_uz_routed, q_lz_trans = runHAPIwithLake(\n", + "st, Sim['Q'], q_uz_routed, q_lz_trans = run_distributed_with_lake(\n", " HBV,\n", " Paths,\n", " ParPath,\n", diff --git a/examples/hydrological-model/Note books/check-colab/Jiboa.ipynb b/examples/hydrological-model/Note books/check-colab/Jiboa.ipynb index 4606a678..509b2fa7 100644 --- a/examples/hydrological-model/Note books/check-colab/Jiboa.ipynb +++ b/examples/hydrological-model/Note books/check-colab/Jiboa.ipynb @@ -287,7 +287,7 @@ "outputs": [], "source": [ "Sim = pd.DataFrame(index=lakeCalib.index)\n", - "st, Sim['Q'], q_uz_routed, q_lz_trans = runHAPIwithLake(\n", + "st, Sim['Q'], q_uz_routed, q_lz_trans = run_distributed_with_lake(\n", " HBV,\n", " Paths,\n", " ParPath,\n", diff --git a/examples/hydrological-model/Note books/coello-distributed-model-run-muskingum.ipynb b/examples/hydrological-model/Note books/coello-distributed-model-run-muskingum.ipynb index dc5a61ec..00d72f28 100644 --- a/examples/hydrological-model/Note books/coello-distributed-model-run-muskingum.ipynb +++ b/examples/hydrological-model/Note books/coello-distributed-model-run-muskingum.ipynb @@ -184,7 +184,7 @@ "metadata": {}, "outputs": [], "source": [ - "Run.RunHapi(Coello)" + "Run.run_distributed(Coello)" ] }, { @@ -196,7 +196,7 @@ "source": [ "import numpy as np\n", "\n", - "np.shape(Coello.Qtot)" + "np.shape(Coello.results.q_total)" ] }, { @@ -248,7 +248,7 @@ "metadata": {}, "outputs": [], "source": [ - "Coello.Qtot[0, 0, 0]" + "Coello.results.q_total[0, 0, 0]" ] }, { @@ -300,7 +300,7 @@ "source": [ "import matplotlib.pyplot as plt\n", "\n", - "plt.plot(Coello.Qtot[12, 1, :])" + "plt.plot(Coello.results.q_total[12, 1, :])" ] }, { diff --git a/examples/hydrological-model/Note books/coello-lumped-model-run-muskingum.ipynb b/examples/hydrological-model/Note books/coello-lumped-model-run-muskingum.ipynb index 03232e66..be7681e1 100644 --- a/examples/hydrological-model/Note books/coello-lumped-model-run-muskingum.ipynb +++ b/examples/hydrological-model/Note books/coello-lumped-model-run-muskingum.ipynb @@ -243,7 +243,7 @@ "metadata": {}, "outputs": [], "source": [ - "Run.runLumped(Coello, Route, RoutingFn)" + "Run.run_lumped(Coello, Route, RoutingFn)" ] }, { diff --git a/examples/hydrological-model/Note books/lumped-model-run-coello.ipynb b/examples/hydrological-model/Note books/lumped-model-run-coello.ipynb index 8d30f9c8..bc64070b 100644 --- a/examples/hydrological-model/Note books/lumped-model-run-coello.ipynb +++ b/examples/hydrological-model/Note books/lumped-model-run-coello.ipynb @@ -229,7 +229,7 @@ "metadata": {}, "outputs": [], "source": [ - "Run.runLumped(Coello, Route, RoutingFn)" + "Run.run_lumped(Coello, Route, RoutingFn)" ] }, { diff --git a/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-NSEHF.py b/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-NSEHF.py index 06ebd692..4dc21069 100644 --- a/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-NSEHF.py +++ b/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-NSEHF.py @@ -95,7 +95,7 @@ def initializer(): def objfn(individual): # Coello.read_parameters(Parameterpath, Snow) Coello.parameters = individual - Run.runLumped(Coello, Route, RoutingFn) + Run.run_lumped(Coello, Route, RoutingFn) # [Coello.QGauges.columns[-1]] NSE = metrics.nse_hf(Coello.QGauges, Coello.Qsim, *Coello.OFArgs) NSEHF = metrics.nse_hf(Coello.QGauges, Coello.Qsim, *Coello.OFArgs) @@ -151,7 +151,7 @@ def distance(individual): # [0.7686518278956287, 144.35510831203874, 1.9922719933560913, 0.1439126168555068, 0.9474744708723734, # 0.749219030317463, 0.8074091462437563, 0.07289588281400794, 68.83482640397304, 5.123384184968337, # 1.9922719933560913] -Run.runLumped(Coello, Route, RoutingFn) +Run.run_lumped(Coello, Route, RoutingFn) ### Calculate Performance Criteria diff --git a/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-RMSE.py b/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-RMSE.py index 1ea8e750..3843bb50 100644 --- a/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-RMSE.py +++ b/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-RMSE.py @@ -97,7 +97,7 @@ def initializer(): def objfn(individual): # Coello.read_parameters(Parameterpath, Snow) Coello.parameters = individual - Run.runLumped(Coello, Route, RoutingFn) + Run.run_lumped(Coello, Route, RoutingFn) # [Coello.QGauges.columns[-1]] NSE = metrics.nse_hf(Coello.QGauges, Coello.Qsim, *Coello.OFArgs) RMSE = metrics.rmse(Coello.QGauges, Coello.Qsim, *Coello.OFArgs) @@ -153,7 +153,7 @@ def distance(individual): # [0.7686518278956287, 144.35510831203874, 1.9922719933560913, 0.1439126168555068, 0.9474744708723734, # 0.749219030317463, 0.8074091462437563, 0.07289588281400794, 68.83482640397304, 5.123384184968337, # 1.9922719933560913] -Run.runLumped(Coello, Route, RoutingFn) +Run.run_lumped(Coello, Route, RoutingFn) ### Calculate Performance Criteria diff --git a/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap.py b/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap.py index a20bf182..146e1bf3 100644 --- a/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap.py +++ b/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap.py @@ -93,7 +93,7 @@ def initializer(): def objfn(individual): # Coello.read_parameters(Parameterpath, Snow) Coello.parameters = individual - Run.runLumped(Coello, Route, RoutingFn) + Run.run_lumped(Coello, Route, RoutingFn) # [Coello.QGauges.columns[-1]] error = PC.NSEHF(Coello.QGauges, Coello.Qsim, *Coello.OFArgs) return (error,) @@ -148,7 +148,7 @@ def distance(individual): # [0.7686518278956287, 144.35510831203874, 1.9922719933560913, 0.1439126168555068, 0.9474744708723734, # 0.749219030317463, 0.8074091462437563, 0.07289588281400794, 68.83482640397304, 5.123384184968337, # 1.9922719933560913] -Run.runLumped(Coello, Route, RoutingFn) +Run.run_lumped(Coello, Route, RoutingFn) ### Calculate Performance Criteria diff --git a/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration.py b/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration.py index 441cb5f5..81983b9b 100644 --- a/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration.py +++ b/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration.py @@ -97,7 +97,7 @@ # %% Run Calibration -cal_parameters = Coello.lumpedCalibration( +cal_parameters = Coello.calibrate_lumped( Basic_inputs, OptimizationArgs, print_error=None ) @@ -107,7 +107,7 @@ # %% Run the Model Coello.parameters = cal_parameters[1] -Run.runLumped(Coello, Route, RoutingFn) +Run.run_lumped(Coello, Route, RoutingFn) ### Calculate Performance Criteria diff --git a/examples/hydrological-model/coello/run/coello-distributed-model-run-maxbas.py b/examples/hydrological-model/coello/run/coello-distributed-model-run-maxbas.py index 20b91922..dd3cc662 100644 --- a/examples/hydrological-model/coello/run/coello-distributed-model-run-maxbas.py +++ b/examples/hydrological-model/coello/run/coello-distributed-model-run-maxbas.py @@ -8,7 +8,7 @@ `python examples/hydrological-model/coello/run/coello-distributed-model-run-maxbas.py`. MAXBAS sends every cell straight to the outlet, so the config loads no flow-direction raster and -`extract_discharge` needs `frame_work_1=True`: a cell of `Qtot` is that cell's contribution to +`extract_discharge` takes the basin-wide sum here: a cell of `q_total` is that cell's contribution to the outlet rather than the discharge at it, which makes the per-gauge shortcut invalid. """ @@ -41,10 +41,10 @@ 6-qlz_translated: [numpy attribute] 3D array of the lower zone discharge translated at each time step """ -Run.runFW1(Coello) +Run.run_maxbas(Coello) # %% calculate performance criteria -Coello.extract_discharge(calculate_metrics=True, frame_work_1=True) +Coello.extract_discharge(calculate_metrics=True) gaugeid = Coello.GaugesTable.loc[Coello.GaugesTable.index[-1], "id"] print("----------------------------------") diff --git a/examples/hydrological-model/coello/run/coello-distributed-model-run-netcdf.py b/examples/hydrological-model/coello/run/coello-distributed-model-run-netcdf.py index 764d1b71..b582106f 100644 --- a/examples/hydrological-model/coello/run/coello-distributed-model-run-netcdf.py +++ b/examples/hydrological-model/coello/run/coello-distributed-model-run-netcdf.py @@ -59,11 +59,11 @@ 6-qlz_translated: [numpy attribute] 3D array of the lower zone discharge translated at each time step """ -Run.RunHapi(Coello) +Run.run_distributed(Coello) # %% Routed fields cover the grid, finite inside the catchment inside = ~np.isnan(Coello.flow_network.flow_acc_arr) -for field_name in ("Qtot", "quz_routed", "qlz_translated"): +for field_name in ("q_total", "quz_routed", "qlz_translated"): field = getattr(Coello, field_name) print( f"{field_name:15s} shape {field.shape}, finite inside: {np.isfinite(field[inside]).all()}" diff --git a/examples/hydrological-model/coello/run/coello-lumped-model-run-maxbas.py b/examples/hydrological-model/coello/run/coello-lumped-model-run-maxbas.py index 224a6ee2..1d9be4a3 100644 --- a/examples/hydrological-model/coello/run/coello-lumped-model-run-maxbas.py +++ b/examples/hydrological-model/coello/run/coello-lumped-model-run-maxbas.py @@ -3,7 +3,7 @@ Everything that used to be a "Paths" block of hardcoded assignments now lives in `coello-lumped-model-run-maxbas.yaml`, next to this script -- `Catchment.from_yaml` reads it and assembles the model. Running it stays here, as in any hand-wired script: the routing function is -a run-time choice rather than an input, so it is picked below and handed to `Run.runLumped`. +a run-time choice rather than an input, so it is picked below and handed to `Run.run_lumped`. The config's `parameters.maxbas: true` says the parameter file carries the triangular-routing parameter; picking `Routing.triangular_routing_1` below is what actually routes with it. @@ -33,7 +33,7 @@ Route = 1 # %% Run the model -Run.runLumped(Coello, Route, RoutingFn) +Run.run_lumped(Coello, Route, RoutingFn) # %% Calculate performance criteria scores = dict() diff --git a/examples/hydrological-model/coello/run/coello-lumped-model-run.py b/examples/hydrological-model/coello/run/coello-lumped-model-run.py index c7671cc3..b3ef8efc 100644 --- a/examples/hydrological-model/coello/run/coello-lumped-model-run.py +++ b/examples/hydrological-model/coello/run/coello-lumped-model-run.py @@ -3,7 +3,7 @@ Everything that used to be a "Paths" block of hardcoded assignments now lives in `coello-lumped-model-run.yaml`, next to this script -- `Catchment.from_yaml` reads it and assembles the model. Running it stays here, as in any hand-wired script: the routing function is -a run-time choice rather than an input, so it is picked below and handed to `Run.runLumped`. +a run-time choice rather than an input, so it is picked below and handed to `Run.run_lumped`. Lumped mode reads one CSV of catchment-average drivers instead of a grid, and one discharge file instead of a gauge table plus a folder -- see the config for both. @@ -33,7 +33,7 @@ Route = 1 # %% Run the model -Run.runLumped(Coello, Route, RoutingFn) +Run.run_lumped(Coello, Route, RoutingFn) # %% Calculate performance criteria scores = {} diff --git a/examples/hydrological-model/jiboa/Jiboa-distributed-model-muskingum-lake.py b/examples/hydrological-model/jiboa/Jiboa-distributed-model-muskingum-lake.py index 9f74d9b3..a5b7603d 100644 --- a/examples/hydrological-model/jiboa/Jiboa-distributed-model-muskingum-lake.py +++ b/examples/hydrological-model/jiboa/Jiboa-distributed-model-muskingum-lake.py @@ -128,9 +128,9 @@ end_date=Date2, ) # %% run the model -Run.runHAPIwithLake(Jiboa, JiboaLake) +Run.run_distributed_with_lake(Jiboa, JiboaLake) # %% calculate some metrics -Jiboa.extract_discharge(only_outlet=True) +Jiboa.extract_discharge() for i in range(len(Jiboa.GaugesTable)): gaugeid = Jiboa.GaugesTable.loc[i, "id"] diff --git a/src/hapi/calibration.py b/src/hapi/calibration.py index 8324d755..987abc0f 100644 --- a/src/hapi/calibration.py +++ b/src/hapi/calibration.py @@ -62,6 +62,12 @@ class Calibration(Catchment): The Calibration class is a subclass of the Catchment superclass, so you need to create the Catchment object first to be able to run the calibration. + + Note: + Results live on `self.results` (a :class:`~hapi.results.SimulationResults`), so the + objective functions read `self.results.q_total` rather than an attribute on the + catchment. Unlike `Run`, this class is still a `Catchment` subclass; converting it + to composition is tracked separately. """ def __init__( @@ -169,35 +175,32 @@ def read_objective_function( def extract_discharge( self, calculate_metrics: bool = True, - frame_work_1: bool = False, factor: list | None = None, - only_outlet: bool = False, ): """Extract the simulated discharge hydrograph at gauge locations. Extracts discharge values from the total routed discharge array - (`self.Qtot`) at each gauge location and stores them in + (`self.results.q_total`) at each gauge location and stores them in `self.Qsim`. Optionally applies a multiplication factor per gauge. Args: calculate_metrics (bool, optional): Whether to calculate performance metrics. Not used in this override but - kept for signature compatibility. Default is True. - frame_work_1 (bool, optional): True if the routing - function is Maxbas. Not used in this override but - kept for signature compatibility. Default is False. + kept so the signature matches the one it overrides. + Default is True. factor (list, optional): List of multiplication factors for the simulated discharge, one per gauge. If None, no scaling is applied. Default is None. - only_outlet (bool, optional): Not used in this override, and inert on the base - class too -- see `Catchment.extract_discharge`. Kept for signature - compatibility. Default is False. + + Raises: + ValueError: The results came from MAXBAS routing, whose per-cell values are + contributions rather than discharges. """ - if self._maxbas_routed: + if not self.results.outlet_shortcut_valid: raise ValueError( "this catchment was run with triangular (MAXBAS) routing, which sends " - "every cell straight to the outlet: a single cell of Qtot is that cell's " + "every cell straight to the outlet: a single cell of q_total is that cell's " "contribution, not the discharge at it, so reading the gauge cells would " "under-report every hydrograph and the objective function would be " "calibrated against the wrong signal." @@ -210,11 +213,13 @@ class too -- see `Catchment.extract_discharge`. Kept for signature Yind = int(self.GaugesTable.loc[self.GaugesTable.index[i], "cell_col"]) # gaugeid = self.GaugesTable.loc[self.GaugesTable.index[i],"id"] - # Quz = self.quz_routed[Xind,Yind,:-1] - # Qlz = self.qlz_translated[Xind,Yind,:-1] + # Quz = self.results.quz_routed[Xind,Yind,:-1] + # Qlz = self.results.qlz_translated[Xind,Yind,:-1] # self.Qsim[:,i] = Quz + Qlz - Qsim = np.reshape(self.Qtot[Xind, Yind, :-1], self.meteo.time_steps) + Qsim = np.reshape( + self.results.q_total[Xind, Yind, :-1], self.meteo.time_steps + ) if factor is not None: self.Qsim[:, i] = Qsim * factor[i] @@ -237,7 +242,7 @@ def run_calibration( Executes the Harmony Search optimization algorithm to calibrate parameters for the conceptual distributed hydrological model. The method distributes parameters spatially using `spatial_var_fun`, - runs the RRM model via `Wrapper.RRMModel`, and evaluates + runs the RRM model via `Wrapper.run_muskingum`, and evaluates performance using the stored objective function. The following attributes must be set on the instance before calling @@ -314,12 +319,12 @@ def opt_fun(par): ) # , kub=spatial_var_fun.Kub, klb=spatial_var_fun.Klb self.parameters = spatial_var_fun.Par3d # run the model - Wrapper.RRMModel(self) + Wrapper.run_muskingum(self) # calculate performance of the model try: error = self.objective_function( self.QGauges, *[self.GaugesTable] - ) # self.qout, self.quz_routed, self.qlz_translated, + ) # self.results.qout, self.results.quz_routed, self.results.qlz_translated, f = list(range(9, len(par), spatial_var_fun.no_parameters)) g = list() for i in range(len(f)): @@ -377,7 +382,7 @@ def opt_fun(par): return res - def FW1Calibration( + def calibrate_maxbas( self, spatial_var_fun: Callable[..., Any], optimization_args: list, @@ -387,7 +392,7 @@ def FW1Calibration( Executes the Harmony Search optimization algorithm to calibrate parameters for the conceptual distributed hydrological model using - the FW1 routing approach via `Wrapper.FW1`. + the FW1 routing approach via `Wrapper.run_maxbas`. The following attributes must be set on the instance before calling this method: @@ -460,11 +465,11 @@ def opt_fun(par): ) # , kub=spatial_var_fun.Kub, klb=spatial_var_fun.Klb, Maskingum=spatial_var_fun.Maskingum self.parameters = spatial_var_fun.Par3d # run the model - Wrapper.FW1(self) + Wrapper.run_maxbas(self) # calculate performance of the model try: error = self.objective_function( - self.QGauges, self.qout, *[self.GaugesTable] + self.QGauges, self.results.qout, *[self.GaugesTable] ) except TypeError as e: # the objective function received fewer inputs than it needs @@ -508,7 +513,7 @@ def opt_fun(par): return res - def lumpedCalibration( + def calibrate_lumped( self, basic_inputs: dict, optimization_args: list, @@ -518,7 +523,7 @@ def lumpedCalibration( Executes the Harmony Search optimization algorithm to calibrate parameters for the lumped conceptual hydrological model. The - method runs the model via `Wrapper.Lumped` and evaluates + method runs the model via `Wrapper.run_lumped` and evaluates performance using the stored objective function. Muskingum routing constraints are enforced as inequality constraints. @@ -594,7 +599,7 @@ def opt_fun(par): # parameters self.parameters = par # run the model - Wrapper.Lumped(self, route, routing_fn) + Wrapper.run_lumped(self, route, routing_fn) # calculate performance of the model try: error = self.objective_function( diff --git a/src/hapi/catchment.py b/src/hapi/catchment.py index 11caf436..7a3702fc 100644 --- a/src/hapi/catchment.py +++ b/src/hapi/catchment.py @@ -71,10 +71,10 @@ } #: Accepted routing methods, mapped to the one spelling the internals compare against. -#: `distrrm.SpatialRouting` tests `routing_method != "Muskingum"` exactly, so the constructor +#: `distrrm.route_muskingum` tests `routing_method != "Muskingum"` exactly, so the constructor #: canonicalises rather than storing what it was handed. `"Kinematic"` belongs here because #: that comparison is also how the flood model selects its own path: a non-Muskingum method -#: with a real `bankfull_depth` skips the cell, which `Run.RunFloodModel` relies on. +#: with a real `bankfull_depth` skips the cell, which `Run.run_flood` relies on. #: `hapi.config.CatchmentConfig.routing_method` exposes the first two to YAML and says why the #: third is not; a method added here needs a decision there too. ROUTING_METHODS = { @@ -218,12 +218,12 @@ class Catchment: The Catchment class includes methods to read the meteorological and spatial inputs of the distributed hydrological model. It also reads the data of the gauges. Build the catchment, then hand it to whichever - :class:`hapi.run.Run` entry point suits it -- `Run.RunHapi(model)`. `Run` states what it + :class:`hapi.run.Run` entry point suits it -- `Run.run_distributed(model)`. `Run` states what it needs as a protocol, which this class satisfies structurally; neither class inherits from the other. A run assigns its output to :attr:`results`. The result arrays are also readable under - their historical names (`Qtot`, `quz`, ...) as read-only properties forwarding to it. + their historical names (`q_total`, `quz`, ...) as read-only properties forwarding to it. """ def __init__( @@ -303,7 +303,7 @@ def __init__( self.date_index = pd.date_range(self.start, self.end, freq="h") # Canonicalised like the two resolutions above, and for a sharper reason: - # `distrrm.SpatialRouting` tests `routing_method != "Muskingum"` case-sensitively, and + # `distrrm.route_muskingum` tests `routing_method != "Muskingum"` case-sensitively, and # the false branch reads `bankfull_depth`, which is None outside the flood model. Left # verbatim, a lower-case "muskingum" therefore routed every cell down the MAXBAS branch # and raised `TypeError: 'NoneType' object is not subscriptable`. @@ -338,7 +338,7 @@ def __init__( self.river_roughness: np.ndarray | None = None self.flood_plain_roughness: np.ndarray | None = None #: Everything one run produced, replaced wholesale by the next run. The seven - #: result arrays below are read-only properties forwarding to it, so `model.Qtot` + #: result arrays below are read-only properties forwarding to it, so `model.results.q_total` #: still reads as it always did while the run layer owns the arrays. `None` until #: a `Run.*` entry point has been called. self.results: SimulationResults | None = None @@ -352,69 +352,6 @@ def __init__( #: path the file already gives. self.config: RunConfig | None = None - # ------------------------------------------------------------------ - # Result accessors - # - # The run layer owns these arrays -- it builds a `SimulationResults` and assigns it to - # `results`. They are exposed here, read-only, under the names they have always had, so - # `Run.RunHapi(model); model.Qtot` reads exactly as before. Read-only on purpose: they - # are outputs, and a run that could be half-overwritten by hand is what the results - # object exists to prevent. To stage a post-run state (a test, say), build a - # `SimulationResults` and assign `model.results`. - # ------------------------------------------------------------------ - - @property - def quz(self) -> np.ndarray | None: - """np.ndarray | None: Upper-zone discharge, or None before a run.""" - return None if self.results is None else self.results.quz - - @property - def qlz(self) -> np.ndarray | None: - """np.ndarray | None: Lower-zone discharge, or None before a run.""" - return None if self.results is None else self.results.qlz - - @property - def state_variables(self) -> np.ndarray | None: - """np.ndarray | None: State array `[sp, sm, uz, lz, wc]`, or None before a run.""" - return None if self.results is None else self.results.state_variables - - @property - def quz_routed(self) -> np.ndarray | None: - """np.ndarray | None: Routed upper-zone discharge, or None before routing.""" - return None if self.results is None else self.results.quz_routed - - @property - def qlz_translated(self) -> np.ndarray | None: - """np.ndarray | None: Translated lower-zone discharge, or None before routing.""" - return None if self.results is None else self.results.qlz_translated - - @property - def Qtot(self) -> np.ndarray | None: - """np.ndarray | None: Total routed discharge, or None before routing. - - How a single cell reads depends on the routing scheme -- see - :attr:`~hapi.results.SimulationResults.outlet_shortcut_valid`. - """ - return None if self.results is None else self.results.Qtot - - @property - def qout(self) -> np.ndarray | None: - """np.ndarray | None: The outlet hydrograph, or None before a run computes one. - - The MAXBAS paths set this during the run; the Muskingum paths leave it for - :meth:`extract_discharge`, which needs the gauge table to find the outlet. - """ - return None if self.results is None else self.results.qout - - @property - def _maxbas_routed(self) -> bool: - """bool: Whether the results came from triangular (MAXBAS) routing. - - Derived from the results rather than tracked as a flag, so it cannot survive into - a later run of a different scheme. - """ - return self.results is not None and not self.results.outlet_shortcut_valid - @classmethod def from_yaml(cls, path: str | Path) -> Self: """Read a YAML run configuration and assemble a model from it. @@ -819,11 +756,11 @@ def read_lumped_inputs(self, path: str): The lumped counterpart of :class:`~hapi.inputs.MeteoInputs`, which carries the distributed drivers: the lumped model works on one column per variable rather than a - grid, and `Wrapper.Lumped` reads the long-term average straight out of the fourth + grid, and `Wrapper.run_lumped` reads the long-term average straight out of the fourth column. A three-column file is completed with a fourth holding the record's mean temperature. - `Wrapper.Lumped` reads that column unconditionally, so without it a file this method + `Wrapper.run_lumped` reads that column unconditionally, so without it a file this method accepts raises `IndexError` in the middle of the run instead. Args: @@ -1179,51 +1116,42 @@ def read_parameters_bound( logger.debug("Parameters' bounds are read successfully") - def extract_discharge( - self, calculate_metrics=True, frame_work_1=False, factor=None, only_outlet=False - ): + def extract_discharge(self, calculate_metrics=True, factor=None): """Extract and sum discharge at gauge locations. - Extracts and sums the discharge from the routed upper zone and - translated lower zone arrays at each gauge location. Optionally - computes performance metrics (RMSE, NSE, NSEhf, KGE, WB, - Pearson-CC, R2) between simulated and observed hydrographs. + Which hydrograph is the right one depends on how the run was routed, and the results + say so, so nothing has to be passed in. Under Muskingum the discharge accumulates + downstream, so each gauge is read from its own cell of `q_total`. Under MAXBAS every + cell is routed straight to the outlet, making a cell that cell's *contribution*; the + hydrograph is then the basin-wide sum the run already computed into `qout`. + + This used to be a `frame_work_1` flag the caller had to set to match the entry point + they had called, with a `ValueError` when they got it wrong. The routing is a + property of the arrays, so it is read off them instead. + + Optionally computes performance metrics (RMSE, NSE, NSEhf, KGE, WB, Pearson-CC, R2) + between the simulated and observed hydrographs. Args: calculate_metrics (bool, optional): Whether to calculate performance metrics. Default is True. - frame_work_1 (bool, optional): True if the routing - function is Maxbas. Default is False. factor (list, optional): List of multiplication factors for simulated discharge at each gauge. Must have the - same length as the number of gauges. Default is None. - only_outlet (bool, optional): Currently has **no effect**. The dispatch below - reads `elif frame_work_1 or only_outlet`, which is reached only when - `frame_work_1` is already True, so this flag never selects anything on its - own. Left in place rather than removed because it is part of the public - signature; pass `frame_work_1=True` for the basin-wide sum. Default is False. + same length as the number of gauges. Applied only on the + per-gauge (Muskingum) path. Default is None. Raises: - ValueError: If the gauge table has not been read yet. + ValueError: The gauge table has not been read, or the model has not been run. """ if self.GaugesTable is None: raise ValueError("please read the gauges' table first.") if self.results is None: raise ValueError( "there are no results to extract; run the model first, e.g. " - "Run.RunHapi(model)" + "Run.run_distributed(model)" ) - if not frame_work_1: - if self._maxbas_routed: - raise ValueError( - "this catchment was run with triangular (MAXBAS) routing, which " - "sends every cell straight to the outlet: a single cell of Qtot is " - "that cell's contribution, not the discharge at it, so reading the " - "outlet cell would under-report the hydrograph. Call " - "extract_discharge(frame_work_1=True) to use the basin-wide sum " - "that Run.runFW1 computed." - ) + if self.results.outlet_shortcut_valid: self.Qsim = pd.DataFrame( index=self.date_index, columns=self.QGauges.columns ) @@ -1234,21 +1162,23 @@ def extract_discharge( outlet_x = self.flow_network.outlet[0][0] outlet_y = self.flow_network.outlet[1][0] - # Muskingum accumulates downstream, so the outlet cell of `Qtot` is the + # Muskingum accumulates downstream, so the outlet cell of `q_total` is the # outlet hydrograph. The engine cannot set this itself: finding the outlet # needs the gauge table, which is an analysis input, not a run input. - self.results.qout = self.Qtot[outlet_x, outlet_y, :] + self.results.qout = self.results.q_total[outlet_x, outlet_y, :] for i in range(len(self.GaugesTable)): x_ind = int(self.GaugesTable.loc[self.GaugesTable.index[i], "cell_row"]) y_ind = int(self.GaugesTable.loc[self.GaugesTable.index[i], "cell_col"]) gauge_id = self.GaugesTable.loc[self.GaugesTable.index[i], "id"] - # Quz = np.reshape(self.quz_routed[x_ind,y_ind,:-1],self.TS-1) - # Qlz = np.reshape(self.qlz_translated[x_ind,y_ind,:-1],self.TS-1) + # Quz = np.reshape(self.results.quz_routed[x_ind,y_ind,:-1],self.TS-1) + # Qlz = np.reshape(self.results.qlz_translated[x_ind,y_ind,:-1],self.TS-1) # q_sim = Quz + Qlz - q_sim = np.reshape(self.Qtot[x_ind, y_ind, :-1], self.meteo.time_steps) + q_sim = np.reshape( + self.results.q_total[x_ind, y_ind, :-1], self.meteo.time_steps + ) if factor is not None: self.Qsim.loc[:, gauge_id] = q_sim * factor[i] else: @@ -1277,10 +1207,12 @@ def extract_discharge( self.metrics.loc["R2", gauge_id] = round( metrics.r2(q_obs, q_sim), 3 ) - elif frame_work_1 or only_outlet: + else: + # MAXBAS: a cell of `q_total` is a contribution, so the hydrograph is the + # basin-wide sum the run already put in `qout`. self.Qsim = pd.DataFrame(index=self.date_index) gauge_id = self.GaugesTable.loc[self.GaugesTable.index[-1], "id"] - q_sim = np.reshape(self.qout, self.meteo.time_steps) + q_sim = np.reshape(self.results.qout, self.meteo.time_steps) self.Qsim.loc[:, gauge_id] = q_sim if calculate_metrics: @@ -1504,28 +1436,28 @@ def plot_distributed_results( end_i = np.nonzero(self.date_index == end)[0][0] if option == 1: - arr = self.Qtot[:, :, start_i:end_i] + arr = self.results.q_total[:, :, start_i:end_i] title = "Total Discharge" elif option == 2: - arr = self.quz_routed[:, :, start_i:end_i] + arr = self.results.quz_routed[:, :, start_i:end_i] title = "Surface Flow" elif option == 3: - arr = self.qlz_translated[:, :, start_i:end_i] + arr = self.results.qlz_translated[:, :, start_i:end_i] title = "Ground Water Flow" elif option == 4: - arr = self.state_variables[:, :, start_i:end_i, 0] + arr = self.results.state_variables[:, :, start_i:end_i, 0] title = "Snow Pack" elif option == 5: - arr = self.state_variables[:, :, start_i:end_i, 1] + arr = self.results.state_variables[:, :, start_i:end_i, 1] title = "Soil Moisture" elif option == 6: - arr = self.state_variables[:, :, start_i:end_i, 2] + arr = self.results.state_variables[:, :, start_i:end_i, 2] title = "Upper Zone" elif option == 7: - arr = self.state_variables[:, :, start_i:end_i, 3] + arr = self.results.state_variables[:, :, start_i:end_i, 3] title = "Lower Zone" elif option == 8: - arr = self.state_variables[:, :, start_i:end_i, 4] + arr = self.results.state_variables[:, :, start_i:end_i, 4] title = "Water Content" elif option == 9: arr = self.meteo.precipitation[:, :, start_i:end_i] @@ -1677,21 +1609,21 @@ def save_results( for i in self.date_index[start_i:end_i] ] if result == 1: - arr = self.Qtot[:, :, start_i:end_i] + arr = self.results.q_total[:, :, start_i:end_i] elif result == 2: - arr = self.quz_routed[:, :, start_i:end_i] + arr = self.results.quz_routed[:, :, start_i:end_i] elif result == 3: - arr = self.qlz_translated[:, :, start_i:end_i] + arr = self.results.qlz_translated[:, :, start_i:end_i] elif result == 4: - arr = self.state_variables[:, :, start_i:end_i, 0] + arr = self.results.state_variables[:, :, start_i:end_i, 0] elif result == 5: - arr = self.state_variables[:, :, start_i:end_i, 1] + arr = self.results.state_variables[:, :, start_i:end_i, 1] elif result == 6: - arr = self.state_variables[:, :, start_i:end_i, 2] + arr = self.results.state_variables[:, :, start_i:end_i, 2] elif result == 7: - arr = self.state_variables[:, :, start_i:end_i, 3] + arr = self.results.state_variables[:, :, start_i:end_i, 3] elif result == 8: - arr = self.state_variables[:, :, start_i:end_i, 4] + arr = self.results.state_variables[:, :, start_i:end_i, 4] else: raise ValueError( f" The result parameter takes a value between 1 and 8, given: {result}" @@ -1714,19 +1646,19 @@ def save_results( data["Qsim"] = self.Qsim[start_i:end_i] data.to_csv(path, index=False, float_format="%.3f") elif result == 2: - data["Quz"] = self.quz[start_i:end_i] + data["Quz"] = self.results.quz[start_i:end_i] data.to_csv(path, index=False, float_format="%.3f") elif result == 3: - data["Qlz"] = self.qlz[start_i:end_i] + data["Qlz"] = self.results.qlz[start_i:end_i] data.to_csv(path, index=False, float_format="%.3f") elif result == 4: - data[STATE_VARIABLES] = self.state_variables[start_i:end_i, :] + data[STATE_VARIABLES] = self.results.state_variables[start_i:end_i, :] data.to_csv(path, index=False, float_format="%.3f") elif result == 5: data["Qsim"] = self.Qsim[start_i:end_i] - data["Quz"] = self.quz[start_i:end_i] - data["Qlz"] = self.qlz[start_i:end_i] - data[STATE_VARIABLES] = self.state_variables[start_i:end_i, :] + data["Quz"] = self.results.quz[start_i:end_i] + data["Qlz"] = self.results.qlz[start_i:end_i] + data[STATE_VARIABLES] = self.results.state_variables[start_i:end_i, :] data.to_csv(path, index=False, float_format="%.3f") else: raise ValueError( diff --git a/src/hapi/config.py b/src/hapi/config.py index 0fea56dc..e5e5e07e 100644 --- a/src/hapi/config.py +++ b/src/hapi/config.py @@ -28,9 +28,9 @@ Out of scope, each because the schema carries no field that reaches it: -- Lake-aware runs (`hapi.catchment.Lake`) and the flood model (`Run.RunFloodModel`), which need +- Lake-aware runs (`hapi.catchment.Lake`) and the flood model (`Run.run_flood`), which need a lake record and a river geometry respectively -- so `read_river_geometry` is unreachable. -- `read_flow_path_length`, and with it `DistMaxbas2`, which scales each cell's MAXBAS by its +- `read_flow_path_length`, and with it `route_maxbas_by_path_length`, which scales each cell's MAXBAS by its distance to the outlet. - Reading a driver folder by numeric file order rather than by date. `MeteoConfig` has no `date` field, so `date=False` can only be reached through `per_variable` -- where it now @@ -506,7 +506,7 @@ def _check_the_routing_method_matches_the_parameter_set(self) -> RunConfig: A lumped run picks its routing function at the call site rather than from this attribute, so `routing_method` is not load-bearing there -- but it is public, it is what - `distrrm.SpatialRouting` keys off, and leaving it saying `Muskingum` on a run using a + `distrrm.route_muskingum` keys off, and leaving it saying `Muskingum` on a run using a MAXBAS parameter set would mislead the next reader. An unstated one is therefore derived from the parameter set rather than left at its default. @@ -570,7 +570,7 @@ def _check_the_distributed_blocks(self) -> None: ) # `flow_direction` is optional on the block because MAXBAS sends every cell straight # to the outlet and never reads one. Muskingum routes along the network, so without - # it the build succeeds and `Run.RunHapi` dereferences a None array after every + # it the build succeeds and `Run.run_distributed` dereferences a None array after every # raster has been read. if ( self.catchment.routing_method == "muskingum" diff --git a/src/hapi/inputs.py b/src/hapi/inputs.py index 31a69b37..1c950a47 100644 --- a/src/hapi/inputs.py +++ b/src/hapi/inputs.py @@ -501,7 +501,7 @@ def __setattr__(self, name: str, value: object) -> None: def acc_val(self) -> list[int]: """list[int]: The distinct accumulation values inside the domain, ascending. - Cached: `SpatialRouting` reads this once per `(accumulation level, row, column)`, so + Cached: `route_muskingum` reads this once per `(accumulation level, row, column)`, so recomputing the `np.unique` on every read costs `(n_acc - 1) x rows x cols` scans of the whole grid -- unnoticeable on the 13x14 test catchment and hours on a real one. Replacing `flow_acc_arr` clears the cache. diff --git a/src/hapi/protocols.py b/src/hapi/protocols.py index fe04750c..7491cc06 100644 --- a/src/hapi/protocols.py +++ b/src/hapi/protocols.py @@ -67,7 +67,7 @@ class DistributedModel(ConceptualModelInputs, Protocol): meteo: The three driver cubes and the calendar they cover. flow_network: The routing network and the grid it defines. date_index: The model's own calendar, which the drivers are checked against. - routing_method: Canonicalised routing method. `SpatialRouting` compares this against + routing_method: Canonicalised routing method. `route_muskingum` compares this against `"Muskingum"` exactly to decide whether a cell is routed or skipped. bankfull_depth: Read only when `routing_method` is not `"Muskingum"`; None otherwise. """ @@ -112,5 +112,3 @@ class FloodModel(DistributedModel, Protocol): river_width: np.ndarray river_roughness: np.ndarray flood_plain_roughness: np.ndarray - - diff --git a/src/hapi/results.py b/src/hapi/results.py index 7c5c36f3..3e6914e8 100644 --- a/src/hapi/results.py +++ b/src/hapi/results.py @@ -8,8 +8,8 @@ the arrays themselves. :class:`SimulationResults` holds them together instead, with the routing scheme as a field. -A catchment exposes the same attribute names as properties forwarding to it, so existing -code reads unchanged; see :class:`~hapi.catchment.Catchment`. +A run assigns one to `Catchment.results`, and that is the only place the arrays live -- read +them as `model.results.q_total`. The catchment carries no result attributes of its own. """ from __future__ import annotations @@ -24,7 +24,7 @@ class RoutingKind(Enum): """Which routing scheme produced a set of results. The distinction is not cosmetic: it decides how a single cell of - :attr:`SimulationResults.Qtot` should be read. Under Muskingum the discharge accumulates + :attr:`SimulationResults.q_total` should be read. Under Muskingum the discharge accumulates downstream, so a cell *is* the discharge at that cell and the outlet cell carries the outlet hydrograph. Under MAXBAS every cell is routed straight to the outlet with its own `maxbas`, so a cell is only that cell's *contribution* and the hydrograph is the sum over @@ -62,7 +62,7 @@ class SimulationResults: `[sp, sm, uz, lz, wc]`. For a lumped run, `(time, 5)`. quz_routed: Upper-zone discharge after routing. `None` until a routing step runs. qlz_translated: Lower-zone discharge after translation. `None` until then. - Qtot: `quz_routed + qlz_translated`. Read it through + q_total: `quz_routed + qlz_translated`. Read it through :attr:`outlet_shortcut_valid` rather than assuming what a cell means. qout: The outlet hydrograph, when the run computed one. The MAXBAS paths sum over the domain and set it directly; the Muskingum paths leave it `None` for @@ -81,7 +81,7 @@ class SimulationResults: ... ) >>> results.routing.value 'unrouted' - >>> results.Qtot is None + >>> results.q_total is None True ``` @@ -107,12 +107,12 @@ class SimulationResults: state_variables: np.ndarray quz_routed: np.ndarray | None = None qlz_translated: np.ndarray | None = None - Qtot: np.ndarray | None = None + q_total: np.ndarray | None = None qout: np.ndarray | None = None @property def outlet_shortcut_valid(self) -> bool: - """bool: Whether a single cell of :attr:`Qtot` is the discharge *at* that cell. + """bool: Whether a single cell of :attr:`q_total` is the discharge *at* that cell. True for every scheme except MAXBAS, which routes each cell straight to the outlet and so makes a cell a contribution rather than a discharge. Reading the outlet cell diff --git a/src/hapi/rrm/distrrm.py b/src/hapi/rrm/distrrm.py index 94ed9a4b..b9684a77 100644 --- a/src/hapi/rrm/distrrm.py +++ b/src/hapi/rrm/distrrm.py @@ -119,7 +119,7 @@ def run_lumped_model(Model) -> SimulationResults: return results @staticmethod - def SpatialRouting(Model): + def route_muskingum(Model): """Route discharge between cells following the flow direction. Accumulates and routes upper-zone discharge (`quz`) using @@ -129,7 +129,7 @@ def SpatialRouting(Model): discharge can be computed at any internal point. After execution the following attributes are set on *Model*: - `quz_routed`, `qlz_translated`, and `Qtot`. + `quz_routed`, `qlz_translated`, and `q_total`. Args: Model (Catchment): A catchment model object carrying the following @@ -234,13 +234,13 @@ def SpatialRouting(Model): results.qlz_translated[x, y, :] = ( results.qlz[x, y, :] + qlzi ) - results.Qtot = results.qlz_translated + results.quz_routed - # Muskingum accumulates downstream, so a cell of `Qtot` is the discharge at that + results.q_total = results.qlz_translated + results.quz_routed + # Muskingum accumulates downstream, so a cell of `q_total` is the discharge at that # cell and the outlet-cell shortcut in `extract_discharge` is valid. results.routing = RoutingKind.MUSKINGUM @staticmethod - def DistMaxbas1(Model): + def route_maxbas(Model): """Route discharge to the outlet using a triangular function. Applies triangular (MAXBAS) routing to the upper-zone @@ -274,10 +274,10 @@ def DistMaxbas1(Model): ) @staticmethod - def DistMaxbas2(Model): + def route_maxbas_by_path_length(Model): """Route discharge using a triangular function scaled by flow path length. - Similar to `DistMaxbas1`, but the MAXBAS parameter for each + Similar to `route_maxbas`, but the MAXBAS parameter for each cell is rescaled proportionally to its flow path length so that cells farther from the outlet receive more attenuation. @@ -322,247 +322,3 @@ def DistMaxbas2(Model): quz[x, y, :] = routing.triangular_routing_2( quz[x, y, :], NormalizedFPL[x, y] ) - - @staticmethod - def Dist_HBV2( - conceptual_model, - lakecell, - q_lake, - DEM, - flow_acc, - flow_acc_plan, - sp_prec, - sp_et, - sp_temp, - sp_pars, - p2, - init_st=None, - ll_temp=None, - q_0=None, - ): - """Run distributed HBV model with lake routing (legacy). - - Executes the HBV conceptual model for every grid cell, routes - lake discharge into the downstream cell using Muskingum - routing, and then routes upper-zone discharge through the - river network. Lower-zone discharge is averaged across all - cells and converted to m3/s. - - Args: - conceptual_model (BaseConceptualModel): Lumped model object with a `simulate` - method. - lakecell (list[int]): Two-element list `[row, col]` - giving the grid indices of the lake cell. - q_lake (numpy.ndarray): 1-D array of lake discharge - time series in m3/s. - DEM (Dataset): pyramids `Dataset` of the catchment DEM. - flow_acc (dict): Flow direction table mapping - `"row,col"` keys to lists of upstream cell index - pairs. - flow_acc_plan (numpy.ndarray): 2-D array of flow - accumulation values; NaN marks no-data cells. - sp_prec (numpy.ndarray): 3-D precipitation array - `(rows, cols, time_steps)`. - sp_et (numpy.ndarray): 3-D evapotranspiration array. - sp_temp (numpy.ndarray): 3-D temperature array. - sp_pars (numpy.ndarray): 3-D parameter array - `(rows, cols, n_params)`. Indices 5, 6, 7 are - K1, K, and alpha; indices 10, 11 are Muskingum - K and X. - p2 (list): Unoptimized parameters. - - - `p2[0]`: tfac -- 1 for hourly, 0.25 for 15 min, - 24 for daily. - - `p2[1]`: Catchment area in km2. - init_st (list, optional): Initial state variable values - `[sp, sm, uz, lz, wc]`. Defaults to None. - ll_temp (numpy.ndarray, optional): 3-D long-term average - temperature array. Defaults to None. - q_0 (float, optional): Initial discharge in m3/s. - Defaults to None. - - Returns: - tuple: A five-element tuple containing: - - - **qout** (*numpy.ndarray*): 1-D discharge time - series at the catchment outlet in m3/s. - - **st** (*numpy.ndarray*): 4-D state variable array - `(rows, cols, time_steps, 5)` with states - `[sp, sm, uz, lz, wc]`. - - **quz_routed** (*numpy.ndarray*): 3-D routed - upper-zone discharge array in m3/s. - - **qlz** (*numpy.ndarray*): 1-D spatially averaged - lower-zone discharge in m3/s. - - **quz** (*numpy.ndarray*): 3-D upper-zone - discharge array in m3/s (before spatial routing). - """ - n_steps = sp_prec.shape[2] + 1 # no of time steps =length of time series +1 - # initialize vector of nans to fill states - dummy_states = np.empty([n_steps, 5]) # [sp,sm,uz,lz,wc] - dummy_states[:] = np.nan - - # Get the mask - no_val = DEM.no_data_value[0] - mask = DEM.read_array(band=0) - # shape of the fpl raster (rows, columns)-------------- rows are x and columns are y - x_ext, y_ext = mask.shape - # y_ext, x_ext = mask.shape # shape of the fpl raster (rows, columns)------------ should change rows are y and columns are x - - # Get deltas of pixel - # get the coordinates of the top left corner and cell size [x,dx,y,dy] - geo_trans = DEM.geotransform - dx = np.abs(geo_trans[1]) / 1000.0 # dx in Km - dy = np.abs(geo_trans[-1]) / 1000.0 # dy in Km - px_area = dx * dy # area of the cell - - # Enumerate the total number of pixels in the catchment - tot_elem = np.sum( - np.sum([[1 for elem in mask_i if elem != no_val] for mask_i in mask]) - ) # get row by row and search [mask_i for mask_i in mask] - - # total pixel area - px_tot_area = tot_elem * px_area # total area of pixels - - # Get number of non-value data - - st = [] # Spatially distributed states - qlz = [] - quz = [] - # ------------------------------------------------------------------------------ - for x in range(x_ext): # no of rows - st_i = [] - q_lzi = [] - q_uzi = [] - # q_out_i = [] - # run all cells in one row ---------------------------------------------------- - for y in range(y_ext): # no of columns - if mask[x, y] != no_val: # only for cells in the domain - # Calculate the states per cell - # TODO optimise for multiprocessing these loops - # _, _st, _uzg, _lzg = conceptual_model.simulate_new_model(avg_prec = sp_prec[x, y,:], - _, _st, _uzg, _lzg = conceptual_model.simulate( - prec=sp_prec[x, y, :], - temp=sp_temp[x, y, :], - et=sp_et[x, y, :], - par=sp_pars[x, y, :], - p2=p2, - init_st=init_st, - ll_temp=None, - q_0=q_0, - snow=0, - ) - # append column after column in the same row ----------------- - st_i.append(np.array(_st)) - # calculate upper zone Q = K1*(LZ_int_1) - q_lz_temp = np.array(sp_pars[x, y, 6]) * _lzg - q_lzi.append(q_lz_temp) - # calculate lower zone Q = k*(UZ_int_3)**(1+alpha) - q_uz_temp = np.array(sp_pars[x, y, 5]) * ( - np.power(_uzg, (1.0 + sp_pars[x, y, 7])) - ) - q_uzi.append(q_uz_temp) - - # print("total = "+str(fff)+"/"+str(tot_elem)+" cell, row= "+str(x+1)+" column= "+str(y+1) ) - else: # if the cell is novalue------------------------------------- - # Fill the empty cells with a nan vector - st_i.append( - dummy_states - ) # fill all states(5 states) for all time steps = nan - q_lzi.append( - dummy_states[:, 0] - ) # q lower zone =nan for all time steps = nan - q_uzi.append( - dummy_states[:, 0] - ) # q upper zone =nan for all time steps = nan - - # store row by row-------- ---------------------------------------------------- - # st.append(st_i) # state variables - st.append(st_i) # state variables - qlz.append(np.array(q_lzi)) # lower zone discharge mm/timestep - quz.append(np.array(q_uzi)) # upper zone routed discharge mm/timestep - # ------------------------------------------------------------------------------ - # convert to arrays - st = np.array(st) # type: ignore - qlz = np.array(qlz) # type: ignore - quz = np.array(quz) # type: ignore - # convert quz from mm/time step to m3/sec - area_coef = p2[1] / px_tot_area - quz = quz * px_area * area_coef / (p2[0] * 3.6) - - no_cells = list( - set( - [ - flow_acc_plan[i, j] - for i in range(x_ext) - for j in range(y_ext) - if not np.isnan(flow_acc_plan[i, j]) - ] - ) - ) - # no_cells=list(set([int(flow_acc_plan[i,j]) for i in range(x_ext) for j in range(y_ext) if flow_acc_plan[i,j] != no_val])) - no_cells.sort() - - # routing lake discharge with DS cell k & x and adding to cell Q - q_lake = routing.muskingum_v( - q_lake, - q_lake[0], - sp_pars[lakecell[0], lakecell[1], 10], - sp_pars[lakecell[0], lakecell[1], 11], - p2[0], - ) - q_lake = np.append(q_lake, q_lake[-1]) - # both lake & Quz are in m3/s - # new - quz[lakecell[0], lakecell[1], :] = quz[lakecell[0], lakecell[1], :] + q_lake - # cells at the divider - quz_routed = np.zeros_like(quz) * np.nan - # for all cell with 0 flow acc put the quz - for x in range(x_ext): # no of rows - for y in range(y_ext): # no of columns - if mask[x, y] != no_val and flow_acc_plan[x, y] == 0: - quz_routed[x, y, :] = quz[x, y, :] - # new - for j in range(1, len(no_cells)): # 2):# - for x in range(x_ext): # no of rows - for y in range(y_ext): # no of columns - # check from total flow accumulation - if mask[x, y] != no_val and flow_acc_plan[x, y] == no_cells[j]: - # print(no_cells[j]) - q_r = np.zeros(n_steps) - for i in range( - len(flow_acc[str(x) + "," + str(y)]) - ): # no_cells[j] - # bring the indexes of the us cell - x_ind = flow_acc[str(x) + "," + str(y)][i][0] - y_ind = flow_acc[str(x) + "," + str(y)][i][1] - # sum the Q of the US cells (already routed for its cell) - # route first with there own k & xthen sum - q_r = q_r + routing.muskingum_v( - quz_routed[x_ind, y_ind, :], - quz_routed[x_ind, y_ind, 0], - sp_pars[x_ind, y_ind, 10], - sp_pars[x_ind, y_ind, 11], - p2[0], - ) - # q=q_r - # add the routed upstream flows to the current Quz in the cell - quz_routed[x, y, :] = quz[x, y, :] + q_r - # check if the max flow _acc is at the outlet - # if tot_elem != np.nanmax(flow_acc_plan): - # raise ("flow accumulation plan is not correct") - # outlet is the cell that has the max flow_acc - outlet = np.where( - flow_acc_plan == np.nanmax(flow_acc_plan) - ) # np.nanmax(flow_acc_plan) - outletx = outlet[0][0] - outlety = outlet[1][0] - - qlz = np.array( # type: ignore[assignment] - [np.nanmean(qlz[:, :, i]) for i in range(n_steps)] # type: ignore[call-overload] - ) # average of all cells (not routed mm/timestep) - # convert Qlz to m3/sec - qlz = qlz * p2[1] / (p2[0] * 3.6) # generation - - qout = qlz + quz_routed[outletx, outlety, :] - - return qout, st, quz_routed, qlz, quz diff --git a/src/hapi/run.py b/src/hapi/run.py index 64ef9d6e..21f302e1 100644 --- a/src/hapi/run.py +++ b/src/hapi/run.py @@ -15,8 +15,7 @@ from __future__ import annotations from collections.abc import Callable -from pathlib import Path -from typing import TYPE_CHECKING, Any, NoReturn +from typing import TYPE_CHECKING, Any import numpy as np import pandas as pd @@ -134,12 +133,12 @@ class Run: afterwards. Methods: - RunHapi: Run the distributed hydrological model. - runHAPIwithLake: Run the distributed model with a lake component. - runFW1: Run the FW1 distributed model. - RunFW1withLake: Run the FW1 model with a lake component. - runLumped: Run the lumped conceptual model. - RunFloodModel: Run the flood model. + run_distributed: Run the distributed hydrological model. + run_distributed_with_lake: Run the distributed model with a lake component. + run_maxbas: Run the FW1 distributed model. + run_maxbas_with_lake: Run the FW1 model with a lake component. + run_lumped: Run the lumped conceptual model. + run_flood: Run the flood model. Examples: - Build a model and run it; the results come back and stay on the model: @@ -150,7 +149,7 @@ class Run: >>> model = Catchment.from_yaml( ... "examples/hydrological-model/coello/run/coello-lumped-model-run.yaml" ... ) - >>> results = Run.runLumped(model, 1, Routing.muskingum_v) + >>> results = Run.run_lumped(model, 1, Routing.muskingum_v) >>> results.routing.value 'lumped' >>> results is model.results @@ -163,42 +162,7 @@ class Run: """ @staticmethod - def from_yaml(path: str | Path) -> NoReturn: - """Refuse to build a `Run`, explaining the pattern instead. - - `Run` is a namespace of entry points, not a model: there is nothing for a - configuration to build. Kept as an explicit refusal because the message it gives is - more useful than the `AttributeError` that would replace it. - - Args: - path: Ignored; present so the call a caller is likely to try is answered. - - Raises: - TypeError: Always. - - Examples: - - The refusal names the pattern to use instead: - ```python - >>> from hapi.run import Run - >>> try: - ... Run.from_yaml("coello-lumped-model-run.yaml") - ... except TypeError as error: - ... print(str(error).split(";")[0]) - Run cannot be built from a configuration - - ``` - - See Also: - hapi.catchment.Catchment.from_yaml: The classmethod that does build a model. - """ - raise TypeError( - "Run cannot be built from a configuration; it holds the entry points that run a " - "model built elsewhere. Build the model with Catchment.from_yaml(path) and pass " - "it in, e.g. Run.RunHapi(model)." - ) - - @staticmethod - def RunHapi(model: DistributedModel) -> SimulationResults: + def run_distributed(model: DistributedModel) -> SimulationResults: """Run the distributed hydrological model. Validates that all input arrays (precipitation, evapotranspiration, @@ -220,7 +184,7 @@ def RunHapi(model: DistributedModel) -> SimulationResults: accumulated and routed at each time step. - `qlz_translated`: 3D array of the lower zone discharge translated at each time step. - - `Qtot`: `quz_routed + qlz_translated`. Routed by Muskingum, so the outlet + - `q_total`: `quz_routed + qlz_translated`. Routed by Muskingum, so the outlet cell carries the outlet hydrograph; `extract_discharge` fills `qout` from it. Raises: @@ -229,13 +193,13 @@ def RunHapi(model: DistributedModel) -> SimulationResults: """ _validate_distributed(model, check_flow_direction=True) # run the model - results = Wrapper.RRMModel(model) + results = Wrapper.run_muskingum(model) logger.info("Model Run has finished") return results @staticmethod - def RunFloodModel(model: FloodModel) -> SimulationResults: + def run_flood(model: FloodModel) -> SimulationResults: """Run the flood model. Runs the conceptual distributed hydrological model with @@ -278,7 +242,7 @@ def RunFloodModel(model: FloodModel) -> SimulationResults: raise ValueError("all input data should have the same number of columns") # run the model - results = Wrapper.RRMModel(model) + results = Wrapper.run_muskingum(model) logger.info("RRM has finished") # SV = SaintVenant() # SV.KinematicRaster(model) @@ -286,7 +250,9 @@ def RunFloodModel(model: FloodModel) -> SimulationResults: return results @staticmethod - def runHAPIwithLake(model: DistributedModel, lake: LakeType) -> SimulationResults: + def run_distributed_with_lake( + model: DistributedModel, lake: LakeType + ) -> SimulationResults: """Run the distributed model with a lake component. Validates that all input arrays have consistent dimensions and @@ -312,13 +278,13 @@ def runHAPIwithLake(model: DistributedModel, lake: LakeType) -> SimulationResult _validate_distributed(model, check_flow_direction=True) _check_lake_meteo(model, lake) # run the model - results = Wrapper.RRMWithlake(model, lake) + results = Wrapper.run_muskingum_with_lake(model, lake) logger.info("Model Run has finished") return results @staticmethod - def runFW1(model: DistributedModel) -> SimulationResults: + def run_maxbas(model: DistributedModel) -> SimulationResults: """Run the FW1 distributed hydrological model. Validates that all input arrays have consistent dimensions, @@ -335,13 +301,13 @@ def runFW1(model: DistributedModel) -> SimulationResults: - `qout`: 1D array of calculated discharge at the catchment outlet, summed over every cell. - `quz`: 3D array of distributed discharge for each cell. - - `Qtot`, `quz_routed`, `qlz_translated`: 3D per-cell fields + - `q_total`, `quz_routed`, `qlz_translated`: 3D per-cell fields read by `save_results` and `plot_distributed_results`. MAXBAS - routes each cell straight to the outlet, so a cell of `Qtot` is + routes each cell straight to the outlet, so a cell of `q_total` is that cell's *contribution* to the outlet — `np.nansum` over the domain reproduces `qout`. Use - `extract_discharge(frame_work_1=True)`; the default outlet-cell - shortcut is invalid for this path and raises. + `extract_discharge` reads the routing off the results and takes the + basin-wide sum on this path automatically. Raises: ValueError: If input data arrays have inconsistent @@ -349,13 +315,15 @@ def runFW1(model: DistributedModel) -> SimulationResults: """ _validate_distributed(model, check_flow_direction=False) # run the model - results = Wrapper.FW1(model) + results = Wrapper.run_maxbas(model) logger.info("Model Run has finished") return results @staticmethod - def RunFW1withLake(model: DistributedModel, lake: LakeType) -> SimulationResults: + def run_maxbas_with_lake( + model: DistributedModel, lake: LakeType + ) -> SimulationResults: """Run the FW1 distributed model with a lake component. Validates that all input arrays have consistent dimensions and @@ -381,10 +349,10 @@ def RunFW1withLake(model: DistributedModel, lake: LakeType) -> SimulationResults _check_lake_meteo(model, lake) # run the model - return Wrapper.FW1Withlake(model, lake) + return Wrapper.run_maxbas_with_lake(model, lake) @staticmethod - def runLumped( + def run_lumped( model: LumpedModelInputs, Route: int = 0, routing_fn: Callable[..., Any] | None = None, @@ -421,7 +389,7 @@ def runLumped( Qsim = pd.DataFrame(index=ind) - results = Wrapper.Lumped(model, Route, routing_fn) + results = Wrapper.run_lumped(model, Route, routing_fn) Qsim["q"] = model.Qsim model.Qsim = Qsim[:] logger.info("Lumped model run has finished successfully") diff --git a/src/hapi/wrapper.py b/src/hapi/wrapper.py index 0f0a8528..44195f5e 100644 --- a/src/hapi/wrapper.py +++ b/src/hapi/wrapper.py @@ -31,11 +31,11 @@ class Wrapper: for Hapi and for FW1 (triangular routing). Methods: - RRMModel: Run distributed RRM with Muskingum spatial routing. - RRMWithlake: Run distributed RRM with lake and Muskingum + run_muskingum: Run distributed RRM with Muskingum spatial routing. + run_muskingum_with_lake: Run distributed RRM with lake and Muskingum spatial routing. FW1: Run distributed RRM with triangular routing. - FW1Withlake: Run distributed RRM with lake and triangular + run_maxbas_with_lake: Run distributed RRM with lake and triangular routing. Lumped: Run a lumped conceptual model with optional routing. """ @@ -45,7 +45,9 @@ def __init__(self): pass @staticmethod - def RRMModel(Model: DistributedModel, ll_temp=None, q_0=None) -> SimulationResults: + def run_muskingum( + Model: DistributedModel, ll_temp=None, q_0=None + ) -> SimulationResults: """Run the distributed rainfall-runoff model with spatial routing. Connects two modules: @@ -86,11 +88,11 @@ def RRMModel(Model: DistributedModel, ll_temp=None, q_0=None) -> SimulationResul # run the GIS part to rout from cell to another. It records # `RoutingKind.MUSKINGUM` on the results, which is what makes the outlet-cell # shortcut in `extract_discharge` valid for them. - distrrm.SpatialRouting(Model) + distrrm.route_muskingum(Model) return results @staticmethod - def RRMWithlake( + def run_muskingum_with_lake( Model: DistributedModel, Lake: Lake, ll_temp=None, q_0=None ) -> SimulationResults: """Run the distributed RRM with lake simulation and routing. @@ -176,16 +178,16 @@ def RRMWithlake( # run the GIS part to rout from cell to another. It records # `RoutingKind.MUSKINGUM` on the results. - distrrm.SpatialRouting(Model) + distrrm.route_muskingum(Model) return results @staticmethod def _set_maxbas_output_fields(Model: ConceptualModelInputs) -> None: """Fill the distributed output fields after a triangular (MAXBAS) run. - `save_results` and `plot_distributed_results` read `Qtot`, + `save_results` and `plot_distributed_results` read `q_total`, `quz_routed` and `qlz_translated` for their discharge options. Only - :meth:`DistRRM.SpatialRouting` (the Muskingum path) used to set them, so + :meth:`DistRRM.route_muskingum` (the Muskingum path) used to set them, so after a MAXBAS run they stayed `None` and every discharge option raised `TypeError: 'NoneType' object is not subscriptable`. @@ -193,9 +195,9 @@ def _set_maxbas_output_fields(Model: ConceptualModelInputs) -> None: cell's own `maxbas`, in place, and applies no cell-to-cell translation to the lower zone. So the routed/translated fields *are* the per-cell arrays, and their sum is the per-cell contribution to the outlet - hydrograph — `np.nansum(Qtot[:, :, i])` reproduces `qout[i]`. That + hydrograph — `np.nansum(q_total[:, :, i])` reproduces `qout[i]`. That differs from the Muskingum path, where the fields accumulate downstream - and `Qtot` at the outlet cell *is* the outlet discharge. + and `q_total` at the outlet cell *is* the outlet discharge. `quz_routed` / `qlz_translated` alias `quz` / `qlz` rather than copying them: they hold the same data, and a copy would double the memory @@ -204,18 +206,20 @@ def _set_maxbas_output_fields(Model: ConceptualModelInputs) -> None: Args: Model: Catchment whose `quz` / `qlz` have been routed by - :meth:`DistRRM.DistMaxbas1`. + :meth:`DistRRM.route_maxbas`. """ results = Model.results results.quz_routed = results.quz results.qlz_translated = results.qlz - results.Qtot = results.qlz + results.quz + results.q_total = results.qlz + results.quz # Marks the outlet-cell shortcut in `extract_discharge` as invalid for these # results, via `SimulationResults.outlet_shortcut_valid`. results.routing = RoutingKind.MAXBAS @staticmethod - def FW1(Model: DistributedModel, ll_temp=None, q_0=None) -> SimulationResults: + def run_maxbas( + Model: DistributedModel, ll_temp=None, q_0=None + ) -> SimulationResults: """Run the distributed RRM with triangular function-1 routing. Connects two modules: @@ -226,7 +230,7 @@ def FW1(Model: DistributedModel, ll_temp=None, q_0=None) -> SimulationResults: The output discharge is computed as the sum of routed upper zone and unrouted lower zone discharge across all cells. - Also fills the per-cell output fields (`Qtot`, `quz_routed`, + Also fills the per-cell output fields (`q_total`, `quz_routed`, `qlz_translated`) via :meth:`_set_maxbas_output_fields`, so the discharge options of `save_results` / `plot_distributed_results` work on this path; see that method for the MAXBAS semantics. @@ -242,7 +246,7 @@ def FW1(Model: DistributedModel, ll_temp=None, q_0=None) -> SimulationResults: # subcatchment results = distrrm.run_lumped_model(Model) - distrrm.DistMaxbas1(Model) + distrrm.route_maxbas(Model) Wrapper._set_maxbas_output_fields(Model) @@ -258,7 +262,7 @@ def FW1(Model: DistributedModel, ll_temp=None, q_0=None) -> SimulationResults: return results @staticmethod - def FW1Withlake( + def run_maxbas_with_lake( Model: DistributedModel, Lake: Lake, ll_temp=None, q_0=None ) -> SimulationResults: """Run the distributed RRM with lake and triangular routing. @@ -325,10 +329,10 @@ def FW1Withlake( # subcatchment results = distrrm.run_lumped_model(Model) - distrrm.DistMaxbas1(Model) + distrrm.route_maxbas(Model) # Subcatchment fields only: the lake is a lumped inflow with no spatial - # extent, so it enters `qout` below but never `Qtot`. + # extent, so it enters `qout` below but never `q_total`. Wrapper._set_maxbas_output_fields(Model) steps = Model.meteo.simulation_steps @@ -350,7 +354,7 @@ def FW1Withlake( return results @staticmethod - def Lumped( + def run_lumped( Model: LumpedModelInputs, Routing: int = 0, RoutingFn: Callable | None = None ) -> SimulationResults: """Run a lumped conceptual model with optional routing. diff --git a/tests/calibration/lumped_calibration.py b/tests/calibration/lumped_calibration.py index b74b12fc..d5176e30 100644 --- a/tests/calibration/lumped_calibration.py +++ b/tests/calibration/lumped_calibration.py @@ -87,7 +87,7 @@ optimization_args = [ApiObjArgs, pll_type, ApiSolveArgs] # %% # run calibration -cal_parameters = Coello.lumpedCalibration( +cal_parameters = Coello.calibrate_lumped( basic_inputs, optimization_args, print_error=None ) @@ -96,7 +96,7 @@ print("Time = " + str(round(cal_parameters[2]["time"] / 60, 2)) + " min") # %% run the model Coello.parameters = cal_parameters[1] -Run.runLumped(Coello, Route, routing_fn) +Run.run_lumped(Coello, Route, routing_fn) # %% calculate performance criteria scores = dict() diff --git a/tests/rrm/calibration/test_calibration_distributed.py b/tests/rrm/calibration/test_calibration_distributed.py index 25424376..0e53a884 100644 --- a/tests/rrm/calibration/test_calibration_distributed.py +++ b/tests/rrm/calibration/test_calibration_distributed.py @@ -69,7 +69,7 @@ def gauged_calibration( Returns: Calibration: Instance carrying `meteo`, `flow_network`, `GaugesTable` and a - synthetic `Qtot` field, with no model run behind it. + synthetic `q_total` field, with no model run behind it. """ coello = Calibration( "coello", @@ -100,7 +100,7 @@ def gauged_calibration( rows, cols = coello.flow_network.rows, coello.flow_network.cols steps = coello.meteo.time_steps rng = np.random.default_rng(1337) - # Stage the post-run state the way the run layer builds it. `Qtot` and the rest are + # Stage the post-run state the way the run layer builds it. `q_total` and the rest are # read-only views onto `results`, so a finished Muskingum run is described rather than # poked in field by field. coello.results = SimulationResults( @@ -108,7 +108,7 @@ def gauged_calibration( quz=np.zeros((rows, cols, steps + 1)), qlz=np.zeros((rows, cols, steps + 1)), state_variables=np.zeros((rows, cols, steps + 1, 5)), - Qtot=rng.random((rows, cols, steps + 1)), + q_total=rng.random((rows, cols, steps + 1)), ) coello.QGauges = DataFrame(rng.random((steps, 2)), columns=[1, 2]) return coello @@ -120,10 +120,10 @@ class TestExtractDischarge: def test_fills_qsim_from_qtot_at_each_gauge_cell( self, gauged_calibration: Calibration ): - """Test that every gauge column is read from its own cell of `Qtot`. + """Test that every gauge column is read from its own cell of `q_total`. Test scenario: - The override reads `Qtot[row, col, :-1]` per gauge and sizes the result from + The override reads `q_total[row, col, :-1]` per gauge and sizes the result from `meteo.time_steps` — the count that moved onto MeteoInputs. Both columns must match the cells the gauge table names, and the trailing step must be dropped. """ @@ -137,13 +137,13 @@ def test_fills_qsim_from_qtot_at_each_gauge_cell( ) np.testing.assert_allclose( coello.Qsim[:, 0], - coello.Qtot[2, 3, :-1], - err_msg="gauge 1 must come from cell (2, 3) of Qtot", + coello.results.q_total[2, 3, :-1], + err_msg="gauge 1 must come from cell (2, 3) of q_total", ) np.testing.assert_allclose( coello.Qsim[:, 1], - coello.Qtot[5, 6, :-1], - err_msg="gauge 2 must come from cell (5, 6) of Qtot", + coello.results.q_total[5, 6, :-1], + err_msg="gauge 2 must come from cell (5, 6) of q_total", ) def test_factor_scales_each_gauge_independently( @@ -161,12 +161,12 @@ def test_factor_scales_each_gauge_independently( np.testing.assert_allclose( coello.Qsim[:, 0], - coello.Qtot[2, 3, :-1] * 2.0, + coello.results.q_total[2, 3, :-1] * 2.0, err_msg="gauge 1 must be scaled by its own factor", ) np.testing.assert_allclose( coello.Qsim[:, 1], - coello.Qtot[5, 6, :-1] * 10.0, + coello.results.q_total[5, 6, :-1] * 10.0, err_msg="gauge 2 must be scaled by its own factor", ) @@ -176,7 +176,7 @@ def test_rejects_a_catchment_routed_with_maxbas( """Test that reading gauge cells after a MAXBAS run raises instead of under-reporting. Test scenario: - Triangular routing sends every cell straight to the outlet, so a cell of `Qtot` + Triangular routing sends every cell straight to the outlet, so a cell of `q_total` is that cell's contribution rather than the discharge at it. Calibrating against it would fit the wrong signal, so the guard must refuse rather than return numbers. """ @@ -278,13 +278,15 @@ def test_the_objective_runs_the_model_on_the_trial_parameters( coello.UB = np.ones(12) ran_with: list[np.ndarray] = [] - original = calibration_module.Wrapper.RRMModel + original = calibration_module.Wrapper.run_muskingum def spy(model, *args, **kwargs): ran_with.append(np.asarray(model.parameters, dtype=float).copy()) return original(model, *args, **kwargs) - monkeypatch.setattr(calibration_module.Wrapper, "RRMModel", staticmethod(spy)) + monkeypatch.setattr( + calibration_module.Wrapper, "run_muskingum", staticmethod(spy) + ) coello.run_calibration(spatial_var_stub, _optimization_args()) @@ -314,7 +316,7 @@ def spy(model, *args, **kwargs): class TestFW1Calibration: - """Tests for `Calibration.FW1Calibration` (triangular routing).""" + """Tests for `Calibration.calibrate_maxbas` (triangular routing).""" def test_stores_the_optimizer_result_on_the_instance( self, gauged_calibration: Calibration, stub_optimizer: dict, spatial_var_stub @@ -330,7 +332,7 @@ def test_stores_the_optimizer_result_on_the_instance( coello.LB = np.zeros(12) coello.UB = np.ones(12) - res = coello.FW1Calibration(spatial_var_stub, _optimization_args()) + res = coello.calibrate_maxbas(spatial_var_stub, _optimization_args()) assert res is CANNED_RESULT, "the optimiser result must be returned untouched" assert coello.OFvalue == pytest.approx(CANNED_RESULT[0]), ( @@ -392,7 +394,7 @@ def test_a_non_dict_bundle_is_refused_before_the_optimizer_is_built( class TestLumpedCalibration: - """Tests for `Calibration.lumpedCalibration`.""" + """Tests for `Calibration.calibrate_lumped`.""" def test_stores_the_optimizer_result_on_the_instance( self, @@ -415,7 +417,7 @@ def test_stores_the_optimizer_result_on_the_instance( Route=0, RoutingFn=Routing.triangular_routing_1, InitialValues=[] ) - res = coello.lumpedCalibration(basic_inputs, _optimization_args()) + res = coello.calibrate_lumped(basic_inputs, _optimization_args()) assert res is CANNED_RESULT, "the optimiser result must be returned untouched" assert coello.OFvalue == pytest.approx(CANNED_RESULT[0]), ( @@ -450,7 +452,7 @@ def test_initial_values_are_seeded_into_the_problem( InitialValues=list(np.full(12, 0.5)), ) - coello.lumpedCalibration(basic_inputs, _optimization_args()) + coello.calibrate_lumped(basic_inputs, _optimization_args()) assert stub_optimizer["n_vars"] == 12, ( f"Expected one variable per bound (12), got {stub_optimizer['n_vars']}" @@ -488,7 +490,7 @@ def test_a_mismatched_initial_values_length_is_refused( optimization_args = _optimization_args() with pytest.raises(ValueError, match="one value per parameter") as exc: - coello.lumpedCalibration(basic_inputs, optimization_args) + coello.calibrate_lumped(basic_inputs, optimization_args) assert "3" in str(exc.value), ( f"the error should name the given length: {exc.value}" diff --git a/tests/rrm/calibration/test_rrm_calibration.py b/tests/rrm/calibration/test_rrm_calibration.py index c21b75be..bd4e99f4 100644 --- a/tests/rrm/calibration/test_rrm_calibration.py +++ b/tests/rrm/calibration/test_rrm_calibration.py @@ -79,7 +79,7 @@ def test_lumped_calibration( optimization_args = [ApiObjArgs, pll_type, ApiSolveArgs] - # cal_parameters = Coello.lumpedCalibration(basic_inputs, optimization_args, print_error=None) + # cal_parameters = Coello.calibrate_lumped(basic_inputs, optimization_args, print_error=None) # assert len(Coello.Qsim) == 1095 and Coello.Qsim.columns.to_list() == ['q'] diff --git a/tests/rrm/catchment/test_config.py b/tests/rrm/catchment/test_config.py index 1757815d..556f02cb 100644 --- a/tests/rrm/catchment/test_config.py +++ b/tests/rrm/catchment/test_config.py @@ -868,7 +868,7 @@ def test_an_unstated_routing_method_is_derived_from_the_parameter_set( Test scenario: The two describe the same choice from opposite sides. Left at its `muskingum` default, a MAXBAS run carried a `routing_method` contradicting what it does -- - and that attribute is what `distrrm.SpatialRouting` keys off. + and that attribute is what `distrrm.route_muskingum` keys off. """ lumped_mapping["parameters"]["maxbas"] = maxbas assert "routing_method" not in lumped_mapping["catchment"], ( @@ -1214,7 +1214,7 @@ def test_any_casing_is_stored_canonically(self, given, stored): stored: The canonical spelling expected on the model. Test scenario: - `distrrm.SpatialRouting` compares `routing_method != "Muskingum"` exactly, and its + `distrrm.route_muskingum` compares `routing_method != "Muskingum"` exactly, and its false branch reads `bankfull_depth`, which is None outside the flood model. A lower-case "muskingum" stored verbatim therefore routed every cell down the MAXBAS branch and raised `TypeError: 'NoneType' object is not subscriptable`. @@ -1229,7 +1229,7 @@ def test_kinematic_is_accepted_for_the_flood_model(self): """Test that the flood model's routing method is still a legal value. Test scenario: - `Run.RunFloodModel` relies on the same `!= "Muskingum"` comparison to skip cells + `Run.run_flood` relies on the same `!= "Muskingum"` comparison to skip cells with a real `bankfull_depth`, so "Kinematic" is a working value and must not be rejected by the new validation. """ @@ -1425,7 +1425,7 @@ def test_the_routing_label_is_the_literal_the_router_compares( tmp_path: pytest temporary directory. Test scenario: - `distrrm.SpatialRouting` tests `routing_method != "Muskingum"` case-sensitively, + `distrrm.route_muskingum` tests `routing_method != "Muskingum"` case-sensitively, and `Catchment.__init__` stores whatever it is given verbatim. A lower-case spelling would send every cell down the MAXBAS branch and read `bankfull_depth`, which is None outside the flood model. @@ -1510,26 +1510,19 @@ def test_the_builder_returns_the_class_it_was_called_on( f"expected a {cls.__name__}, got {type(model).__name__}" ) - def test_run_cannot_be_built_because_it_takes_no_constructor_arguments( - self, distributed_mapping, tmp_path - ): - """Test that `Run.from_yaml` fails loudly rather than building something unusable. - - Args: - distributed_mapping: A complete distributed configuration. - tmp_path: pytest temporary directory. + def test_run_has_no_constructor_and_nothing_to_build_from_a_configuration(self): + """Test that `Run` is a namespace of entry points, not something a config builds. Test scenario: - `Run` inherits the classmethod but overrides `__init__` to take only `self`, and - its entry points are called unbound on a catchment (`Run.RunHapi(model)`). The - override refuses the call with a message naming that pattern, rather than letting - constructor arity produce a `TypeError` about an unexpected keyword argument -- - an error that says nothing about what to do instead. + `Run` used to inherit `Catchment.from_yaml` through the subclassing, so the call + resolved and had to be overridden to refuse. It no longer inherits anything, so + the name is simply absent -- which is the honest answer for a class that holds + no state and models nothing. """ - path = write_yaml(distributed_mapping, tmp_path) - - with pytest.raises(TypeError, match="Catchment.from_yaml"): - Run.from_yaml(path) + assert not hasattr(Run, "from_yaml"), ( + "Run must not offer from_yaml; a configuration builds a Catchment, which is " + "then passed to an entry point" + ) def test_a_lumped_configuration_reads_the_averaged_driver_csv( self, lumped_mapping, tmp_path diff --git a/tests/rrm/catchment/test_e2e_coello_from_netcdf.py b/tests/rrm/catchment/test_e2e_coello_from_netcdf.py index d0dd5fbf..74df504f 100644 --- a/tests/rrm/catchment/test_e2e_coello_from_netcdf.py +++ b/tests/rrm/catchment/test_e2e_coello_from_netcdf.py @@ -152,7 +152,7 @@ def muskingum_run(meteo_from_one_file: MeteoInputs, setup: dict) -> Catchment: maxbas=False, with_flow_direction=True, ) - Run.RunHapi(model) + Run.run_distributed(model) return model @@ -175,7 +175,7 @@ def maxbas_run( maxbas=True, with_flow_direction=False, ) - Run.runFW1(model) + Run.run_maxbas(model) return model @@ -215,7 +215,7 @@ def test_routing_fills_the_distributed_fields(self, muskingum_run: Catchment): """Test that the run populates the per-cell output fields at grid size. Test scenario: - `Qtot`, `quz_routed` and `qlz_translated` back every downstream reader -- + `q_total`, `quz_routed` and `qlz_translated` back every downstream reader -- `extract_discharge`, `save_results`, the animations. All three must come back at `(rows, cols, simulation_steps)` and finite inside the catchment. """ @@ -224,8 +224,8 @@ def test_routing_fills_the_distributed_fields(self, muskingum_run: Catchment): steps = model.meteo.simulation_steps inside = ~np.isnan(model.flow_network.flow_acc_arr) - for name in ("Qtot", "quz_routed", "qlz_translated"): - field = getattr(model, name) + for name in ("q_total", "quz_routed", "qlz_translated"): + field = getattr(model.results, name) assert field is not None, f"{name} must be set by the run" assert field.shape == (rows, cols, steps), ( f"{name} should be {(rows, cols, steps)}, got {field.shape}" @@ -239,7 +239,7 @@ def test_gauge_extraction_and_metrics(self, muskingum_run: Catchment): Test scenario: The end of the chain a modeller actually reads. `extract_discharge` walks the - gauge table, pulls each gauge's cell out of `Qtot`, and scores it against the + gauge table, pulls each gauge's cell out of `q_total`, and scores it against the observations -- so this is where a driver that never reached the model, or reached it shifted in time, would finally show up as a non-finite score. """ @@ -269,12 +269,12 @@ def test_gauge_extraction_and_metrics(self, muskingum_run: Catchment): def test_saved_rasters_carry_the_routed_discharge( self, muskingum_run: Catchment, coello_acc_path: str, tmp_path ): - """Test that the results reach disk as readable rasters holding `Qtot`. + """Test that the results reach disk as readable rasters holding `q_total`. Test scenario: The last link, and the one nothing else exercises for the NetCDF-driven path: `save_results` writes one raster per step, georeferenced from the flow - accumulation grid. Reading the first one back and comparing it to `Qtot`'s first + accumulation grid. Reading the first one back and comparing it to `q_total`'s first slice proves the file holds the run's own numbers rather than an empty grid. """ model = muskingum_run @@ -290,7 +290,7 @@ def test_saved_rasters_carry_the_routed_discharge( ) first = Dataset.read_file(str(written[0])).read_array() - expected = model.Qtot[:, :, 0] + expected = model.results.q_total[:, :, 0] inside = ~np.isnan(model.flow_network.flow_acc_arr) np.testing.assert_allclose( np.asarray(first)[inside], @@ -318,8 +318,8 @@ def test_runs_without_a_flow_direction_raster(self, maxbas_run: Catchment): assert model.flow_network.has_flow_direction is False, ( "the fixture must load accumulation only, or this proves nothing" ) - assert model.Qtot is not None, "the triangular run must fill Qtot" - assert model._maxbas_routed is True, ( + assert model.results.q_total is not None, "the triangular run must fill q_total" + assert not model.results.outlet_shortcut_valid, ( "the triangular path must mark the model, so extract_discharge refuses the " "outlet-cell shortcut" ) @@ -328,17 +328,14 @@ def test_basin_wide_discharge_and_metrics(self, maxbas_run: Catchment): """Test that the MAXBAS run scores against the gauges via the basin-wide sum. Test scenario: - Triangular routing makes a cell of `Qtot` a contribution rather than a discharge, - so the per-gauge shortcut is refused and `frame_work_1=True` selects the + Triangular routing makes a cell of `q_total` a contribution rather than a discharge, + so the per-gauge shortcut is skipped and the routing kind selects the basin-wide sum instead. That is the only way to score this path, and it has to produce the same seven finite metrics. """ model = maxbas_run - with pytest.raises(ValueError, match="MAXBAS"): - model.extract_discharge(calculate_metrics=False) - - model.extract_discharge(calculate_metrics=True, frame_work_1=True) + model.extract_discharge(calculate_metrics=True) assert isinstance(model.metrics, DataFrame), ( f"metrics should be a DataFrame, got {type(model.metrics)}" @@ -374,7 +371,7 @@ def test_the_netcdf_run_matches_the_raster_run_cell_for_cell( maxbas=False, with_flow_direction=True, ) - Run.RunHapi(raster_model) + Run.run_distributed(raster_model) for name in METEO_VARIABLES: np.testing.assert_array_equal( @@ -383,8 +380,8 @@ def test_the_netcdf_run_matches_the_raster_run_cell_for_cell( err_msg=f"{name} differs before the run even starts", ) np.testing.assert_allclose( - muskingum_run.Qtot, - raster_model.Qtot, + muskingum_run.results.q_total, + raster_model.results.q_total, rtol=1e-9, err_msg="the routed discharge field differs between the two sources", ) diff --git a/tests/rrm/catchment/test_extract_discharge_distributed.py b/tests/rrm/catchment/test_extract_discharge_distributed.py index 387daee9..856e17af 100644 --- a/tests/rrm/catchment/test_extract_discharge_distributed.py +++ b/tests/rrm/catchment/test_extract_discharge_distributed.py @@ -28,7 +28,7 @@ def coello_muskingum_run( """Distributed Coello catchment with a completed Muskingum run. Returns: - Catchment: Model with `Qtot` populated by the spatial routing. + Catchment: Model with `q_total` populated by the spatial routing. """ coello = Catchment( "coello", @@ -52,7 +52,7 @@ def coello_muskingum_run( coello.read_lumped_model(HBVLumped, coello_cat_area, coello_initial_cond) coello.read_gauge_table(coello_gauges_table, coello_acc_path) coello.read_discharge_gauges(coello_gauges_path, column="id", fmt="%Y-%m-%d") - Run.RunHapi(coello) + Run.run_distributed(coello) return coello @@ -61,8 +61,8 @@ def test_extract_discharge_distributed_metrics(coello_muskingum_run: Catchment): Test scenario: After a Muskingum run, extract_discharge with the default - frame_work_1=False walks the gauge table, extracts Qsim per gauge - from Qtot, and fills the metrics frame (RMSE, NSE, NSEhf, KGE, WB, + Muskingum-routed results walk the gauge table, extracting Qsim per gauge + from q_total, and fills the metrics frame (RMSE, NSE, NSEhf, KGE, WB, Pearson-CC, R2) with finite numbers. """ coello = coello_muskingum_run diff --git a/tests/rrm/catchment/test_flow_network.py b/tests/rrm/catchment/test_flow_network.py index 295b5450..bbba5859 100644 --- a/tests/rrm/catchment/test_flow_network.py +++ b/tests/rrm/catchment/test_flow_network.py @@ -308,7 +308,7 @@ def test_repeated_reads_return_the_same_object(self, network: FlowNetwork): """Test that `acc_val` is computed once rather than on every read. Test scenario: - `SpatialRouting` reads this once per (accumulation level, row, column), so a + `route_muskingum` reads this once per (accumulation level, row, column), so a property that reruns `np.unique` over the whole grid turns the routing loop into `(n_acc - 1) x rows x cols` full-grid scans. On the 13x14 test catchment that is invisible; on a 100x100 catchment it dominates the run. Identity across reads is diff --git a/tests/rrm/catchment/test_fw1_output_fields.py b/tests/rrm/catchment/test_fw1_output_fields.py index 4ff1f3f3..dbd8f55e 100644 --- a/tests/rrm/catchment/test_fw1_output_fields.py +++ b/tests/rrm/catchment/test_fw1_output_fields.py @@ -1,7 +1,7 @@ """Tests for the per-cell output fields the triangular (MAXBAS) path produces. -Only ``DistRRM.SpatialRouting`` (the Muskingum path) used to set ``Qtot`` / -``quz_routed`` / ``qlz_translated``, so after ``Run.runFW1`` they stayed ``None`` and every +Only ``DistRRM.route_muskingum`` (the Muskingum path) used to set ``q_total`` / +``quz_routed`` / ``qlz_translated``, so after ``Run.run_maxbas`` they stayed ``None`` and every discharge option of ``save_results`` / ``plot_distributed_results`` raised ``TypeError: 'NoneType' object is not subscriptable``. ``Wrapper._set_maxbas_output_fields`` now fills them; these tests pin both the values and the MAXBAS-specific semantics. @@ -60,7 +60,7 @@ def coello_fw1( # read here rather than inside a test: the fixture is module-scoped, so a test that # mutated it would leak into whichever test ran next. coello.read_gauge_table(coello_gauges_table, coello_acc_path) - Run.runFW1(coello) + Run.run_maxbas(coello) return coello @@ -107,7 +107,7 @@ def coello_unrouted( def test_fw1_sets_the_per_cell_output_fields(coello_fw1: Catchment): - """Test that runFW1 leaves Qtot and the routed/translated fields populated. + """Test that run_maxbas leaves q_total and the routed/translated fields populated. Args: coello_fw1: Coello catchment with a completed MAXBAS run. @@ -117,30 +117,30 @@ def test_fw1_sets_the_per_cell_output_fields(coello_fw1: Catchment): `plot_distributed_results`. Before the fix only the Muskingum path set them, so they were `None` here and every discharge option raised. """ - shape = coello_fw1.quz.shape - for name in ("Qtot", "quz_routed", "qlz_translated"): - field = getattr(coello_fw1, name) - assert field is not None, f"{name} must be set after runFW1" + shape = coello_fw1.results.quz.shape + for name in ("q_total", "quz_routed", "qlz_translated"): + field = getattr(coello_fw1.results, name) + assert field is not None, f"{name} must be set after run_maxbas" assert field.shape == shape, f"{name} must be a per-cell, per-timestep field" def test_fw1_qtot_matches_an_independent_triangular_convolution( coello_fw1: Catchment, coello_unrouted: Catchment ): - """Test `Qtot` against the routing recomputed outside the wrapper. + """Test `q_total` against the routing recomputed outside the wrapper. Args: coello_fw1: Coello catchment with a completed MAXBAS run. coello_unrouted: The same catchment with only the per-cell model run. Test scenario: - Asserting `Qtot == qlz + quz` restates the assignment that produced it, so a wrong + Asserting `q_total == qlz + quz` restates the assignment that produced it, so a wrong MAXBAS convolution passes. Recompute the routing here instead -- each in-domain cell convolved with `triangular_routing_1` against its own MAXBAS parameter -- and compare the whole field. That is the only assertion that can tell a correct kernel from a wrong one. """ - expected_quz = coello_unrouted.quz.copy() + expected_quz = coello_unrouted.results.quz.copy() maxbas = coello_fw1.parameters[:, :, -1] acc = coello_fw1.flow_network.flow_acc_arr for x in range(coello_fw1.flow_network.rows): @@ -151,12 +151,12 @@ def test_fw1_qtot_matches_an_independent_triangular_convolution( ) np.testing.assert_allclose( - coello_fw1.Qtot, - coello_unrouted.qlz + expected_quz, + coello_fw1.results.q_total, + coello_unrouted.results.qlz + expected_quz, rtol=1e-6, - err_msg="Qtot must be the lower zone plus the independently routed upper zone", + err_msg="q_total must be the lower zone plus the independently routed upper zone", ) - assert not np.allclose(expected_quz, coello_unrouted.quz), ( + assert not np.allclose(expected_quz, coello_unrouted.results.quz), ( "the routing must change quz, otherwise this comparison proves nothing" ) @@ -171,46 +171,50 @@ def test_fw1_routes_the_upper_zone_but_leaves_the_lower_zone_alone( coello_unrouted: The same catchment with only the per-cell model run. Test scenario: - `DistMaxbas1` convolves only the upper zone. Pinning both halves separates a routing + `route_maxbas` convolves only the upper zone. Pinning both halves separates a routing that ran from one that silently did nothing, and catches a change that started routing the lower zone too. """ np.testing.assert_allclose( - coello_fw1.qlz, - coello_unrouted.qlz, + coello_fw1.results.qlz, + coello_unrouted.results.qlz, rtol=1e-9, err_msg="the lower zone is not routed by the triangular path", ) - assert not np.allclose(coello_fw1.quz, coello_unrouted.quz), ( + assert not np.allclose(coello_fw1.results.quz, coello_unrouted.results.quz), ( "the upper zone must be attenuated by the triangular routing" ) -def test_extract_discharge_rejects_the_outlet_cell_shortcut_after_fw1( - coello_fw1: Catchment, -): - """Test that the Muskingum-only outlet-cell shortcut raises on a MAXBAS run. +def test_extract_discharge_takes_the_basin_wide_sum_after_fw1(coello_fw1: Catchment): + """Test that MAXBAS results select the basin-wide hydrograph without being told. Args: coello_fw1: Coello catchment with a completed MAXBAS run. - coello_acc_path: Flow-accumulation raster, to map the gauges onto the grid. - coello_gauges_table: Gauge table path. Test scenario: - `extract_discharge(frame_work_1=False)` reads `Qtot` at the outlet cell. - That is correct for Muskingum, where the field accumulates downstream, but - wrong for MAXBAS, where one cell holds only its own contribution. Before - Qtot was populated this crashed with a bare TypeError; now that it holds - real numbers the wrong answer would be silent, so it must raise instead. + Reading `q_total` at the outlet cell is right for Muskingum, where the field + accumulates downstream, and wrong for MAXBAS, where a cell holds only its own + contribution. This used to be the caller's job via a `frame_work_1` flag, with a + `ValueError` when they set it wrong. The routing is a property of the arrays, so + `extract_discharge` reads it off the results and takes the sum the run computed. """ - with pytest.raises(ValueError, match="MAXBAS"): - coello_fw1.extract_discharge(calculate_metrics=False) + coello_fw1.extract_discharge(calculate_metrics=False) + + assert coello_fw1.Qsim.shape[1] == 1, ( + f"the basin-wide path yields one hydrograph, got {coello_fw1.Qsim.shape[1]} columns" + ) + np.testing.assert_allclose( + coello_fw1.Qsim.iloc[:, 0].to_numpy(dtype="float64"), + np.asarray(coello_fw1.results.qout, dtype="float64"), + err_msg="the extracted hydrograph must be the qout the MAXBAS run summed", + ) def test_save_results_distributed_discharge_after_fw1( coello_fw1: Catchment, coello_acc_path: str, tmp_path ): - """Test that the discharge results can now be written as rasters after runFW1. + """Test that the discharge results can now be written as rasters after run_maxbas. Args: coello_fw1: Coello catchment with a completed MAXBAS run. @@ -220,7 +224,7 @@ def test_save_results_distributed_discharge_after_fw1( Test scenario: `result=1` (total discharge) is the option that used to raise on this path. Writes it and reads the first raster back to confirm it carries the - matching Qtot slice. + matching q_total slice. """ out = tmp_path / "q" out.mkdir() @@ -238,23 +242,23 @@ def test_save_results_distributed_discharge_after_fw1( start_i = np.where(coello_fw1.date_index == np.datetime64("2009-01-01"))[0][0] np.testing.assert_allclose( Dataset.read_file(str(written[0])).read_array(band=0), - coello_fw1.Qtot[:, :, start_i], + coello_fw1.results.q_total[:, :, start_i], rtol=1e-5, - err_msg="the first raster must hold the first Qtot step", + err_msg="the first raster must hold the first q_total step", ) @pytest.mark.plot @pytest.mark.parametrize("option", [1, 2, 3]) def test_plot_discharge_options_after_fw1(coello_fw1: Catchment, option: int): - """Test that the three discharge animation options work after runFW1. + """Test that the three discharge animation options work after run_maxbas. Args: coello_fw1: Coello catchment with a completed MAXBAS run. option: 1 total discharge, 2 upper zone, 3 ground water. Test scenario: - Options 1-3 read Qtot / quz_routed / qlz_translated respectively and all + Options 1-3 read q_total / quz_routed / qlz_translated respectively and all three raised TypeError on this path before the fix. """ import matplotlib.animation diff --git a/tests/rrm/catchment/test_maxbas_routing_variants.py b/tests/rrm/catchment/test_maxbas_routing_variants.py index 07275a9a..fe521dfa 100644 --- a/tests/rrm/catchment/test_maxbas_routing_variants.py +++ b/tests/rrm/catchment/test_maxbas_routing_variants.py @@ -1,8 +1,8 @@ """Tests for the two triangular-routing variants and the lumped routing branches. -`DistributedRRM.DistMaxbas2` is a public entry point that nothing inside the package calls -- it +`DistributedRRM.route_maxbas_by_path_length` is a public entry point that nothing inside the package calls -- it rescales each cell's MAXBAS by its flow path length -- so it is reached only from a test or a -downstream user. `Wrapper.Lumped` picks its routing call by whether `maxbas` is set, a branch +downstream user. `Wrapper.run_lumped` picks its routing call by whether `maxbas` is set, a branch the existing lumped tests never took. """ @@ -89,7 +89,7 @@ def maxbas_parameters_path(lumped_parameters_path: str, tmp_path_factory) -> str class TestDistMaxbas2: - """Tests for `DistributedRRM.DistMaxbas2`.""" + """Tests for `DistributedRRM.route_maxbas_by_path_length`.""" def test_conserves_volume_while_redistributing_it_in_time( self, coello_before_routing: Catchment @@ -103,19 +103,19 @@ def test_conserves_volume_while_redistributing_it_in_time( an equality. """ model = coello_before_routing - before = model.quz.copy() + before = model.results.quz.copy() inside = ~np.isnan(model.flow_path_length_arr) - DistributedRRM.DistMaxbas2(model) + DistributedRRM.route_maxbas_by_path_length(model) - assert not np.array_equal(model.quz[inside], before[inside]), ( + assert not np.array_equal(model.results.quz[inside], before[inside]), ( "the routing must alter the upper-zone discharge of the masked cells" ) - assert np.isfinite(model.quz[inside]).all(), ( + assert np.isfinite(model.results.quz[inside]).all(), ( "routed discharge must stay finite inside the catchment" ) np.testing.assert_allclose( - np.nansum(model.quz[inside]), + np.nansum(model.results.quz[inside]), np.nansum(before[inside]), rtol=1e-6, err_msg="triangular routing must conserve volume, not merely bound it", @@ -128,7 +128,7 @@ def test_attenuates_the_peak_of_a_cell_far_from_the_outlet( Test scenario: This variant exists to give distant cells more attenuation than near ones -- that - is the whole difference from `DistMaxbas1`. Comparing the peak reduction of the + is the whole difference from `route_maxbas`. Comparing the peak reduction of the nearest and furthest in-domain cells is what distinguishes it from a uniform kernel; a routing that ignored the flow path would attenuate both equally. """ @@ -141,12 +141,12 @@ def test_attenuates_the_peak_of_a_cell_far_from_the_outlet( furthest = np.unravel_index( np.nanargmax(np.where(inside, fpl, np.nan)), fpl.shape ) - before = model.quz.copy() + before = model.results.quz.copy() - DistributedRRM.DistMaxbas2(model) + DistributedRRM.route_maxbas_by_path_length(model) - near_drop = before[nearest].max() - model.quz[nearest].max() - far_drop = before[furthest].max() - model.quz[furthest].max() + near_drop = before[nearest].max() - model.results.quz[nearest].max() + far_drop = before[furthest].max() - model.results.quz[furthest].max() assert far_drop > near_drop, ( f"the cell {far_drop:.4g} from the outlet must be attenuated more than the near " f"one ({near_drop:.4g}); equal attenuation means the flow path was ignored" @@ -165,14 +165,14 @@ def test_leaves_cells_outside_the_mask_untouched( """ model = coello_before_routing outside = np.isnan(model.flow_path_length_arr) - sentinel = np.linspace(1.0, 10.0, model.quz.shape[2]) - model.quz[outside] = sentinel - before = model.quz.copy() + sentinel = np.linspace(1.0, 10.0, model.results.quz.shape[2]) + model.results.quz[outside] = sentinel + before = model.results.quz.copy() - DistributedRRM.DistMaxbas2(model) + DistributedRRM.route_maxbas_by_path_length(model) np.testing.assert_array_equal( - model.quz[outside], + model.results.quz[outside], before[outside], err_msg="cells with no flow-path length must not be routed", ) @@ -185,7 +185,7 @@ def _lumped_model( area: float, initial_cond: list, ) -> Catchment: - """Build a lumped catchment ready for `Run.runLumped`. + """Build a lumped catchment ready for `Run.run_lumped`. Args: dates: `[start, end]` simulation dates. @@ -205,7 +205,7 @@ def _lumped_model( class TestLumpedRouting: - """Tests for the routing branches of `Wrapper.Lumped` reached through `Run.runLumped`.""" + """Tests for the routing branches of `Wrapper.run_lumped` reached through `Run.run_lumped`.""" def test_maxbas_routing_convolves_qsim_with_the_last_parameter( self, @@ -218,13 +218,13 @@ def test_maxbas_routing_convolves_qsim_with_the_last_parameter( """Test that the MAXBAS branch routes on the trailing parameter and nothing else. Test scenario: - `Wrapper.Lumped` has two routing calls: the MAXBAS one passes a single parameter, + `Wrapper.run_lumped` has two routing calls: the MAXBAS one passes a single parameter, the Muskingum one passes three. Which runs is decided by the `maxbas` flag `read_parameters` stored. Asserting only that `Qsim` is finite would pass for an unrouted series, so compare against the same run left unrouted, convolved independently with the parameter the branch is supposed to use. """ - # Straight to the wrapper: `Run.runLumped` wraps `Qsim` in a date-indexed frame and + # Straight to the wrapper: `Run.run_lumped` wraps `Qsim` in a date-indexed frame and # the unrouted series is one step longer than the index, so only the routed form # survives that call. unrouted = _lumped_model( @@ -234,7 +234,7 @@ def test_maxbas_routing_convolves_qsim_with_the_last_parameter( coello_AreaCoeff, coello_InitialCond, ) - Wrapper.Lumped(unrouted, Routing=0) + Wrapper.run_lumped(unrouted, Routing=0) routed = _lumped_model( coello_rrm_date, lumped_meteo_data_path, @@ -243,13 +243,13 @@ def test_maxbas_routing_convolves_qsim_with_the_last_parameter( coello_InitialCond, ) - Run.runLumped(routed, Route=1, routing_fn=Routing.triangular_routing_1) + Run.run_lumped(routed, Route=1, routing_fn=Routing.triangular_routing_1) maxbas = routed.parameters[-1] expected = Routing.triangular_routing_1( np.array(np.asarray(unrouted.Qsim)[:-1]), maxbas ) - # `runLumped` wraps the routed series in a date-indexed frame; compare the values. + # `run_lumped` wraps the routed series in a date-indexed frame; compare the values. actual = np.asarray(routed.Qsim, dtype=float).ravel() np.testing.assert_allclose( actual, @@ -285,7 +285,7 @@ def test_routing_without_a_function_is_rejected( model.read_parameters(lumped_parameters_path, False, maxbas=False) with pytest.raises(ValueError, match="routing_fn"): - Run.runLumped(model, Route=1) + Run.run_lumped(model, Route=1) class TestCalculateWeightsGuard: @@ -309,7 +309,7 @@ def test_a_maxbas_below_one_is_refused(self, maxbas, symptom): Test scenario: `triangular_routing_2` and the three conceptual models already carried this - check; `calculate_weights` -- which `triangular_routing_1` and `DistMaxbas1` + check; `calculate_weights` -- which `triangular_routing_1` and `route_maxbas` resolve their weights through -- did not, and each value failed differently or not at all. """ @@ -324,7 +324,7 @@ def test_the_guard_reaches_triangular_routing_1(self): """Test that the routing function itself refuses, not only the weights helper. Test scenario: - `triangular_routing_1` is what the lumped MAXBAS example and `DistMaxbas1` call, + `triangular_routing_1` is what the lumped MAXBAS example and `route_maxbas` call, and it delegates to `calculate_weights` on its first line -- so the guard has to surface there rather than being swallowed. """ diff --git a/tests/rrm/catchment/test_meteo_inputs.py b/tests/rrm/catchment/test_meteo_inputs.py index fce26194..1d460fee 100644 --- a/tests/rrm/catchment/test_meteo_inputs.py +++ b/tests/rrm/catchment/test_meteo_inputs.py @@ -90,7 +90,7 @@ def _run(model_name: str, inputs: MeteoInputs, fixtures: dict) -> Catchment: coello.read_lumped_model(HBVLumped, fixtures["area"], fixtures["initial"]) coello.read_gauge_table(fixtures["gauges_table"], fixtures["acc"]) coello.read_discharge_gauges(fixtures["gauges"], column="id", fmt="%Y-%m-%d") - Run.RunHapi(coello) + Run.run_distributed(coello) return coello @@ -1023,15 +1023,15 @@ def test_netcdf_run_reproduces_the_raster_run( Test scenario: The end-to-end claim: swapping folders of rasters for NetCDFs must not move the - hydrograph by a single cell. Compares the routed `Qtot` field and the per-gauge + hydrograph by a single cell. Compares the routed `q_total` field and the per-gauge `Qsim` extracted from it. """ raster_run = _run("coello-rasters", from_rasters, fixtures) netcdf_run = _run("coello-netcdf", from_netcdf_files, fixtures) np.testing.assert_allclose( - netcdf_run.Qtot, - raster_run.Qtot, + netcdf_run.results.q_total, + raster_run.results.q_total, rtol=1e-9, err_msg="the routed discharge field differs between the two sources", ) @@ -1055,7 +1055,7 @@ def coello_muskingum_from_netcdf( raster reader calls replaced by a single `MeteoInputs.from_netcdf` load of `meteo.nc`. Returns: - Catchment: Model with `Qtot` populated by the spatial routing. + Catchment: Model with `q_total` populated by the spatial routing. """ return _run("coello-combined-netcdf", from_combined_netcdf, fixtures) @@ -1112,15 +1112,15 @@ def test_reproduces_the_raster_run( Test scenario: Packing all three drivers into one file, with the caller naming which variable is - which, must not move the hydrograph. Compares the routed `Qtot` field and the + which, must not move the hydrograph. Compares the routed `q_total` field and the per-gauge `Qsim` against a run fed from the raster folders. """ raster_run = _run("coello-rasters-vs-combined", from_rasters, fixtures) combined = coello_muskingum_from_netcdf np.testing.assert_allclose( - combined.Qtot, - raster_run.Qtot, + combined.results.q_total, + raster_run.results.q_total, rtol=1e-9, err_msg="the routed discharge field differs from the raster-driven run", ) diff --git a/tests/rrm/catchment/test_plot_animation.py b/tests/rrm/catchment/test_plot_animation.py index 97b03132..dac92bb5 100644 --- a/tests/rrm/catchment/test_plot_animation.py +++ b/tests/rrm/catchment/test_plot_animation.py @@ -44,7 +44,7 @@ def coello_animated( coello.read_parameters(coello_dist_parameters_maxbas, False, maxbas=True) coello.read_lumped_model(HBVLumped, coello_cat_area, coello_initial_cond) coello.read_gauge_table(coello_gauges_table, coello_acc_path) - Run.runFW1(coello) + Run.run_maxbas(coello) return coello @@ -67,12 +67,14 @@ def test_plot_state_variable(coello_animated: Catchment): """Animating a state variable after a model run works.""" import matplotlib.animation - before = coello_animated.state_variables.copy() + before = coello_animated.results.state_variables.copy() anim = coello_animated.plot_distributed_results( "2009-01-01", "2009-01-09", option=5 ) assert isinstance(anim, matplotlib.animation.FuncAnimation) - assert np.array_equal(before, coello_animated.state_variables, equal_nan=True) + assert np.array_equal( + before, coello_animated.results.state_variables, equal_nan=True + ) @pytest.mark.plot diff --git a/tests/rrm/catchment/test_read_parameters_validation.py b/tests/rrm/catchment/test_read_parameters_validation.py index b06d01bb..a2b8e1fd 100644 --- a/tests/rrm/catchment/test_read_parameters_validation.py +++ b/tests/rrm/catchment/test_read_parameters_validation.py @@ -411,7 +411,7 @@ def test_a_three_column_file_gains_the_long_term_average( """Test that the fourth column is derived rather than left missing. Test scenario: - The method documents 3 or 4 columns, but `Wrapper.Lumped` reads `data[:, 3]` + The method documents 3 or 4 columns, but `Wrapper.run_lumped` reads `data[:, 3]` unconditionally. A three-column file was therefore accepted here and then raised `IndexError` in the middle of the run. The derived column is the record's mean temperature, which is what the reader this replaced computed. diff --git a/tests/rrm/catchment/test_rrm_catchment.py b/tests/rrm/catchment/test_rrm_catchment.py index 73600c47..5e04d731 100644 --- a/tests/rrm/catchment/test_rrm_catchment.py +++ b/tests/rrm/catchment/test_rrm_catchment.py @@ -82,7 +82,7 @@ def test_run_lumped( coello.read_discharge_gauges(lumped_gauges_path, fmt=coello_gauges_date_fmt) routing_fn = Routing.muskingum_v route = 1 - Run.runLumped(coello, route, routing_fn) + Run.run_lumped(coello, route, routing_fn) assert len(coello.Qsim) == 10 assert coello.Qsim.columns.to_list() == ["q"] @@ -108,7 +108,7 @@ def test_save_lumped_results( # discharge gauges coello.read_discharge_gauges(lumped_gauges_path, fmt=coello_gauges_date_fmt) Route = 1 - Run.runLumped(coello, Route, Routing.muskingum_v) + Run.run_lumped(coello, Route, Routing.muskingum_v) coello.save_results(result=5, path=path) # # TODO: still not finished as it does not run the plotHydrograph method @@ -131,7 +131,7 @@ def test_save_lumped_results( # coello.readDischargeGauges(lumped_gauges_path, fmt=coello_gauges_date_fmt) # RoutingFn = Routing.muskingum_v # Route = 1 - # Run.runLumped(coello, Route, RoutingFn) + # Run.run_lumped(coello, Route, RoutingFn) # assert len(coello.Qsim) == 10 and coello.Qsim.columns.to_list() == ["q"] @@ -358,12 +358,17 @@ def test_run_dist( # coello.readFlowDir(coello_fd_path) coello.read_parameters(coello_dist_parameters_maxbas, False, maxbas=True) coello.read_lumped_model(HBVLumped, coello_cat_area, coello_initial_cond) - Run.runFW1(coello) - assert isinstance(coello.qout, np.ndarray) - assert len(coello.qout) == 10 - assert coello.state_variables.shape == (coello_shape[0], coello_shape[1], 11, 5) - assert coello.quz.shape == (coello_shape[0], coello_shape[1], 11) - assert coello.qlz.shape == (coello_shape[0], coello_shape[1], 11) + Run.run_maxbas(coello) + assert isinstance(coello.results.qout, np.ndarray) + assert len(coello.results.qout) == 10 + assert coello.results.state_variables.shape == ( + coello_shape[0], + coello_shape[1], + 11, + 5, + ) + assert coello.results.quz.shape == (coello_shape[0], coello_shape[1], 11) + assert coello.results.qlz.shape == (coello_shape[0], coello_shape[1], 11) def test_extract_results( self, @@ -408,9 +413,9 @@ def test_extract_results( snow = False coello.read_parameters(coello_dist_parameters_maxbas, snow, maxbas=True) coello.read_lumped_model(HBVLumped, coello_cat_area, coello_initial_cond) - Run.runFW1(coello) + Run.run_maxbas(coello) - coello.extract_discharge(calculate_metrics=True, frame_work_1=True) + coello.extract_discharge(calculate_metrics=True) assert isinstance(coello.metrics, DataFrame) assert len(coello.metrics) == 7 assert len(coello.Qsim) == 10 @@ -419,7 +424,7 @@ def test_extract_results( class TestSaveAndExtractAfterFW1: """A second FW1 run covering `save_results` and `extract_discharge`. - Named for what it does: it reads the MAXBAS parameter set and calls `Run.runFW1`, so + Named for what it does: it reads the MAXBAS parameter set and calls `Run.run_maxbas`, so calling it `TestMuskingum` said the opposite of what it exercises. The Muskingum path is covered by `test_extract_discharge_distributed.py` and `test_meteo_inputs.py`. """ @@ -461,12 +466,17 @@ def test_run_dist( Snow = False coello.read_parameters(coello_dist_parameters_maxbas, Snow, maxbas=True) coello.read_lumped_model(HBVLumped, coello_cat_area, coello_initial_cond) - Run.runFW1(coello) - assert isinstance(coello.qout, np.ndarray) - assert len(coello.qout) == 10 - assert coello.state_variables.shape == (coello_shape[0], coello_shape[1], 11, 5) - assert coello.quz.shape == (coello_shape[0], coello_shape[1], 11) - assert coello.qlz.shape == (coello_shape[0], coello_shape[1], 11) + Run.run_maxbas(coello) + assert isinstance(coello.results.qout, np.ndarray) + assert len(coello.results.qout) == 10 + assert coello.results.state_variables.shape == ( + coello_shape[0], + coello_shape[1], + 11, + 5, + ) + assert coello.results.quz.shape == (coello_shape[0], coello_shape[1], 11) + assert coello.results.qlz.shape == (coello_shape[0], coello_shape[1], 11) def test_extract_results( self, @@ -511,9 +521,9 @@ def test_extract_results( Snow = False coello.read_parameters(coello_dist_parameters_maxbas, Snow, maxbas=True) coello.read_lumped_model(HBVLumped, coello_cat_area, coello_initial_cond) - Run.runFW1(coello) + Run.run_maxbas(coello) - coello.extract_discharge(calculate_metrics=True, frame_work_1=True) + coello.extract_discharge(calculate_metrics=True) assert isinstance(coello.metrics, DataFrame) assert len(coello.metrics) == 7 assert len(coello.Qsim) == 10 diff --git a/tests/rrm/catchment/test_run_results_coupling.py b/tests/rrm/catchment/test_run_results_coupling.py index e9b7510e..203366ec 100644 --- a/tests/rrm/catchment/test_run_results_coupling.py +++ b/tests/rrm/catchment/test_run_results_coupling.py @@ -1,6 +1,6 @@ """Tests for how `Run` and `Catchment` are coupled, and for the results object between them. -`Run` used to subclass `Catchment` and be called unbound (`Run.RunHapi(model)`, with the +`Run` used to subclass `Catchment` and be called unbound (`Run.run_distributed(model)`, with the catchment landing on `self`), writing nine result arrays back onto the model plus a private `_maxbas_routed` flag recording which routing had produced them. It is now a namespace of static entry points that state what they need as a protocol and return a @@ -34,7 +34,7 @@ "state_variables", "quz_routed", "qlz_translated", - "Qtot", + "q_total", "qout", ) @@ -120,13 +120,12 @@ def test_run_does_not_subclass_catchment(self): @pytest.mark.parametrize( "name", [ - "RunHapi", - "RunFloodModel", - "runHAPIwithLake", - "runFW1", - "RunFW1withLake", - "runLumped", - "from_yaml", + "run_distributed", + "run_flood", + "run_distributed_with_lake", + "run_maxbas", + "run_maxbas_with_lake", + "run_lumped", ], ) def test_every_entry_point_is_a_static_method(self, name: str): @@ -174,15 +173,26 @@ def test_run_does_not_import_catchment_at_runtime(self): "hapi.protocols exist so the run layer does not depend on the concrete class" ) - def test_from_yaml_refuses_and_names_the_pattern(self): - """Test that `Run.from_yaml` explains itself rather than raising AttributeError. + def test_run_offers_nothing_a_configuration_could_build(self): + """Test that `Run` exposes no constructor-like surface. Test scenario: - `Run` no longer inherits `Catchment.from_yaml`, so the call would fail with a - bare AttributeError. The explicit refusal is kept because it names what to do. + The old inheritance made `Run.from_yaml` resolve to `Catchment.from_yaml`, which + had to be overridden to refuse. With the inheritance gone the name is absent, + and `Run` carries only the entry points. """ - with pytest.raises(TypeError, match="cannot be built from a configuration"): - Run.from_yaml("anything.yaml") + assert not hasattr(Run, "from_yaml"), ( + "Run must not offer from_yaml; build a Catchment and pass it to an entry point" + ) + public = {n for n in vars(Run) if not n.startswith("_")} + assert public == { + "run_distributed", + "run_distributed_with_lake", + "run_maxbas", + "run_maxbas_with_lake", + "run_lumped", + "run_flood", + }, f"Run should carry only the six entry points, got {sorted(public)}" class TestEntryPointsReturnTheirResults: @@ -200,16 +210,16 @@ def test_run_hapi_returns_the_object_it_put_on_the_model( """ model = _build("coello", coello_dist_parameters_muskingum, **coello_fixtures) - results = Run.RunHapi(model) + results = Run.run_distributed(model) assert isinstance(results, SimulationResults), ( - f"RunHapi must return SimulationResults, got {type(results).__name__}" + f"run_distributed must return SimulationResults, got {type(results).__name__}" ) assert results is model.results, ( "the returned results must be the same object assigned to model.results" ) assert results.routing is RoutingKind.MUSKINGUM, ( - f"a RunHapi run is Muskingum-routed, got {results.routing}" + f"a run_distributed run is Muskingum-routed, got {results.routing}" ) def test_run_fw1_returns_maxbas_routed_results( @@ -225,79 +235,81 @@ def test_run_fw1_returns_maxbas_routed_results( "coello", coello_dist_parameters_maxbas, maxbas=True, **coello_fixtures ) - results = Run.runFW1(model) + results = Run.run_maxbas(model) assert results.routing is RoutingKind.MAXBAS, ( - f"a runFW1 run is MAXBAS-routed, got {results.routing}" + f"a run_maxbas run is MAXBAS-routed, got {results.routing}" ) assert not results.outlet_shortcut_valid, ( "MAXBAS sends every cell to the outlet, so the outlet-cell shortcut is invalid" ) -class TestResultAttributesAreReadOnlyViews: - """The historical attribute names still read, but the results object owns the arrays.""" +class TestResultsAreTheOnlyHomeForTheArrays: + """The catchment carries no result attributes; `results` is where they live.""" @pytest.mark.parametrize("field", RESULT_FIELDS) - def test_field_reads_through_to_the_results_object( - self, coello_fixtures: dict, coello_dist_parameters_muskingum: str, field: str - ): - """Test that each historical name returns exactly what the results object holds. + def test_the_catchment_does_not_carry_the_field(self, field: str): + """Test that a result name is absent from the catchment entirely. Test scenario: - Existing scripts and notebooks read `model.Qtot` and friends after a run. The - properties exist so that keeps working; identity is asserted rather than - equality so a copy cannot pass. + These were nullable attributes on `Catchment`, then briefly properties + forwarding to `results`. Both are gone: there is one home for the arrays, so a + reader cannot pick the stale one by habit. Args: - field: The result attribute being checked. + field: The result field being checked. """ - model = _build("coello", coello_dist_parameters_muskingum, **coello_fixtures) - Run.RunHapi(model) + model = Catchment("empty", "2009-01-01", "2009-01-10") - assert getattr(model, field) is getattr(model.results, field), ( - f"model.{field} must read through to results.{field}, not copy it" + assert not hasattr(model, field), ( + f"Catchment must not carry {field}; read it as model.results.{field}" ) @pytest.mark.parametrize("field", RESULT_FIELDS) - def test_field_is_none_before_a_run(self, field: str): - """Test that the result names read as None on a catchment that has not run. + def test_the_results_object_carries_the_field_after_a_run( + self, coello_fixtures: dict, coello_dist_parameters_muskingum: str, field: str + ): + """Test that a completed Muskingum run populates every result field. Test scenario: - They used to be None-initialised attributes. Reading one before a run must stay - a None rather than becoming an AttributeError. + The Muskingum path fills all of them except `qout`, which needs the gauge table + and so is left for `extract_discharge`. Everything else must be an array. Args: - field: The result attribute being checked. + field: The result field being checked. """ - model = Catchment("empty", "2009-01-01", "2009-01-10") + model = _build("coello", coello_dist_parameters_muskingum, **coello_fixtures) + results = Run.run_distributed(model) - assert getattr(model, field) is None, ( - f"model.{field} must be None before a run, got {type(getattr(model, field))}" - ) + value = getattr(results, field) + if field == "qout": + assert value is None, ( + "the Muskingum path leaves qout for extract_discharge to fill" + ) + else: + assert isinstance(value, np.ndarray), ( + f"results.{field} must be an array after a run, got {type(value).__name__}" + ) - @pytest.mark.parametrize("field", RESULT_FIELDS) - def test_field_cannot_be_assigned(self, field: str): - """Test that a result field rejects assignment. + def test_results_is_none_before_a_run(self): + """Test that a catchment that has not run has no results at all. Test scenario: - These are outputs. A run that can be half-overwritten by hand is exactly what - the results object exists to prevent, so the properties have no setter and - staging a state goes through `model.results` instead. - - Args: - field: The result attribute being checked. + One `None` to check instead of nine, which is the point: a model either has a + finished run behind it or it does not. """ model = Catchment("empty", "2009-01-01", "2009-01-10") - with pytest.raises(AttributeError): - setattr(model, field, np.zeros((2, 2, 2))) + assert model.results is None, ( + f"a catchment that has not run must have results=None, got {model.results}" + ) -class TestRoutingProvenanceReplacesTheFlag: - """`_maxbas_routed` is derived from the results, so it cannot outlive the run.""" +class TestRoutingProvenanceTravelsWithTheArrays: + """Routing is a field of the results, so it cannot outlive the run that set it.""" - def test_a_muskingum_run_after_a_maxbas_run_clears_the_maxbas_reading( + def test_a_muskingum_run_after_a_maxbas_run_reads_as_muskingum( self, coello_fixtures: dict, coello_dist_parameters_maxbas: str, @@ -306,41 +318,34 @@ def test_a_muskingum_run_after_a_maxbas_run_clears_the_maxbas_reading( """Test that running MAXBAS then Muskingum leaves the model reading as Muskingum. Test scenario: - This is the case the old boolean needed hand-clearing for: `_maxbas_routed` was - set by the MAXBAS path and had to be reset by every Muskingum path, with a - comment saying so. Deriving it from the results makes that impossible to forget, - because a new run replaces the object the reading comes from. + This is the case the old boolean needed hand-clearing for: it was set by the + MAXBAS path and had to be reset by every Muskingum path, with a comment saying + so. Carrying the routing on the results makes that impossible to forget, because + a new run replaces the object the reading comes from. """ model = _build( "coello", coello_dist_parameters_maxbas, maxbas=True, **coello_fixtures ) - Run.runFW1(model) - assert model._maxbas_routed, "the FW1 run should read as MAXBAS-routed" + maxbas_results = Run.run_maxbas(model) + assert maxbas_results.routing is RoutingKind.MAXBAS, ( + "the MAXBAS run should record its own routing" + ) # Re-read the parameters the Muskingum path needs, then run it on the same model. model.read_parameters(coello_dist_parameters_muskingum, False) - Run.RunHapi(model) + muskingum_results = Run.run_distributed(model) - assert not model._maxbas_routed, ( - "after a Muskingum run the model must no longer read as MAXBAS-routed" + assert muskingum_results is not maxbas_results, ( + "a second run must build a new results object, not overwrite fields in place" + ) + assert model.results.routing is RoutingKind.MUSKINGUM, ( + f"after a Muskingum run the model must read as Muskingum, got " + f"{model.results.routing}" ) assert model.results.outlet_shortcut_valid, ( "the outlet-cell shortcut is valid again once Muskingum has routed the results" ) - def test_a_fresh_catchment_does_not_read_as_maxbas_routed(self): - """Test that a model that has never run does not claim MAXBAS routing. - - Test scenario: - The derived reading has to answer for the no-results case too, since - `extract_discharge` consults it before checking anything else. - """ - model = Catchment("empty", "2009-01-01", "2009-01-10") - - assert not model._maxbas_routed, ( - "a catchment with no results must not read as MAXBAS-routed" - ) - class TestValidationNamesWhatIsMissing: """The guards the protocols exposed: an optional input that this path does require.""" @@ -359,7 +364,7 @@ def test_a_muskingum_run_without_a_flow_direction_raster_says_so( model = _build("coello", coello_dist_parameters_muskingum, **fixtures) with pytest.raises(ValueError, match="flow-direction raster"): - Run.RunHapi(model) + Run.run_distributed(model) def test_the_flood_model_names_the_river_geometry_it_lacks( self, coello_fixtures: dict, coello_dist_parameters_muskingum: str @@ -373,7 +378,7 @@ def test_the_flood_model_names_the_river_geometry_it_lacks( model = _build("coello", coello_dist_parameters_muskingum, **coello_fixtures) with pytest.raises(ValueError, match="read_river_geometry") as exc_info: - Run.RunFloodModel(model) + Run.run_flood(model) assert "bankfull_depth" in str(exc_info.value), ( f"the error should name the missing rasters, got: {exc_info.value}" diff --git a/tests/rrm/catchment/test_run_validation.py b/tests/rrm/catchment/test_run_validation.py index 8c54f6c9..2ce6de00 100644 --- a/tests/rrm/catchment/test_run_validation.py +++ b/tests/rrm/catchment/test_run_validation.py @@ -1,6 +1,6 @@ """Tests for the input validation `Run`'s lake and flood entry points perform. -These three entry points (`RunFloodModel`, `runHAPIwithLake`, `RunFW1withLake`) each open with a +These three entry points (`run_flood`, `run_distributed_with_lake`, `run_maxbas_with_lake`) each open with a block of dimension checks that now read through `flow_network` and `meteo` rather than through attributes on the catchment. The wrapper each one dispatches to is replaced by a spy, so the tests pin the validation and the dispatch without running a lake or a hydraulic model. @@ -43,7 +43,7 @@ def _spy(*args): return _spy - for name in ("RRMModel", "RRMWithlake", "FW1Withlake"): + for name in ("run_muskingum", "run_muskingum_with_lake", "run_maxbas_with_lake"): monkeypatch.setattr(run_module.Wrapper, name, staticmethod(_make(name))) return calls @@ -103,7 +103,7 @@ def _load_flat_river_geometry(model: Catchment) -> None: class TestRunFloodModel: - """Tests for `Run.RunFloodModel`.""" + """Tests for `Run.run_flood`.""" def test_dispatches_once_every_input_lines_up( self, coello_loaded: Catchment, spied_wrapper: dict @@ -113,14 +113,16 @@ def test_dispatches_once_every_input_lines_up( Test scenario: The flood entry point checks the flow-direction grid, the meteo cubes, the parameter array and the four river-geometry rasters before dispatching. With all - of them on the catchment grid it must call `Wrapper.RRMModel` with the model. + of them on the catchment grid it must call `Wrapper.run_muskingum` with the model. """ _load_flat_river_geometry(coello_loaded) - Run.RunFloodModel(coello_loaded) + Run.run_flood(coello_loaded) - assert "RRMModel" in spied_wrapper, "RRMModel should have been dispatched" - assert spied_wrapper["RRMModel"][0] is coello_loaded, ( + assert "run_muskingum" in spied_wrapper, ( + "run_muskingum should have been dispatched" + ) + assert spied_wrapper["run_muskingum"][0] is coello_loaded, ( "the wrapper must receive the model itself" ) @@ -138,9 +140,9 @@ def test_rejects_river_geometry_off_the_catchment_grid( coello_loaded.river_width = coello_loaded.river_width[:-1, :] with pytest.raises(ValueError, match="number of rows"): - Run.RunFloodModel(coello_loaded) + Run.run_flood(coello_loaded) - assert "RRMModel" not in spied_wrapper, ( + assert "run_muskingum" not in spied_wrapper, ( "the wrapper must not run on inconsistent geometry" ) @@ -163,15 +165,15 @@ def test_rejects_meteo_that_does_not_cover_the_grid( ) with pytest.raises(ValueError, match="must share the catchment's grid"): - Run.RunFloodModel(coello_loaded) + Run.run_flood(coello_loaded) - assert "RRMModel" not in spied_wrapper, ( + assert "run_muskingum" not in spied_wrapper, ( "the wrapper must not run on inconsistent meteo inputs" ) class TestRunHapiWithLake: - """Tests for `Run.runHAPIwithLake`.""" + """Tests for `Run.run_distributed_with_lake`.""" def test_dispatches_once_the_lake_record_matches_the_simulation( self, coello_loaded: Catchment, spied_wrapper: dict @@ -181,14 +183,16 @@ def test_dispatches_once_the_lake_record_matches_the_simulation( Test scenario: The lake is a lumped inflow whose own meteorological record must run step for step with the distributed cubes. A matching record must dispatch to - `Wrapper.RRMWithlake` with both the model and the lake. + `Wrapper.run_muskingum_with_lake` with both the model and the lake. """ lake = _LakeStub(coello_loaded.meteo.time_steps) - Run.runHAPIwithLake(coello_loaded, lake) + Run.run_distributed_with_lake(coello_loaded, lake) - assert "RRMWithlake" in spied_wrapper, "RRMWithlake should have been dispatched" - assert spied_wrapper["RRMWithlake"] == (coello_loaded, lake), ( + assert "run_muskingum_with_lake" in spied_wrapper, ( + "run_muskingum_with_lake should have been dispatched" + ) + assert spied_wrapper["run_muskingum_with_lake"] == (coello_loaded, lake), ( "the wrapper must receive the model and the lake" ) @@ -206,9 +210,9 @@ def test_rejects_a_lake_record_of_the_wrong_length( lake = _LakeStub(coello_loaded.meteo.time_steps - 1) with pytest.raises(ValueError, match="same length"): - Run.runHAPIwithLake(coello_loaded, lake) + Run.run_distributed_with_lake(coello_loaded, lake) - assert "RRMWithlake" not in spied_wrapper, ( + assert "run_muskingum_with_lake" not in spied_wrapper, ( "the wrapper must not run against a mismatched lake record" ) @@ -224,7 +228,7 @@ def test_rejects_a_lake_record_missing_a_column( lake = _LakeStub(coello_loaded.meteo.time_steps, columns=2) with pytest.raises(ValueError, match="three columns"): - Run.runHAPIwithLake(coello_loaded, lake) + Run.run_distributed_with_lake(coello_loaded, lake) def test_rejects_a_flow_direction_grid_of_the_wrong_shape( self, coello_loaded: Catchment, spied_wrapper: dict @@ -242,11 +246,11 @@ def test_rejects_a_flow_direction_grid_of_the_wrong_shape( lake = _LakeStub(coello_loaded.meteo.time_steps) with pytest.raises(ValueError, match="rows and columns"): - Run.runHAPIwithLake(coello_loaded, lake) + Run.run_distributed_with_lake(coello_loaded, lake) class TestRunFW1WithLake: - """Tests for `Run.RunFW1withLake`.""" + """Tests for `Run.run_maxbas_with_lake`.""" def test_dispatches_once_the_lake_record_matches_the_simulation( self, coello_loaded: Catchment, spied_wrapper: dict @@ -256,14 +260,16 @@ def test_dispatches_once_the_lake_record_matches_the_simulation( Test scenario: The FW1 lake path skips the flow-direction check — triangular routing needs no direction grid — but keeps the meteo, parameter and lake checks. A consistent - model must reach `Wrapper.FW1Withlake`. + model must reach `Wrapper.run_maxbas_with_lake`. """ lake = _LakeStub(coello_loaded.meteo.time_steps) - Run.RunFW1withLake(coello_loaded, lake) + Run.run_maxbas_with_lake(coello_loaded, lake) - assert "FW1Withlake" in spied_wrapper, "FW1Withlake should have been dispatched" - assert spied_wrapper["FW1Withlake"] == (coello_loaded, lake), ( + assert "run_maxbas_with_lake" in spied_wrapper, ( + "run_maxbas_with_lake should have been dispatched" + ) + assert spied_wrapper["run_maxbas_with_lake"] == (coello_loaded, lake), ( "the wrapper must receive the model and the lake" ) @@ -280,8 +286,8 @@ def test_rejects_parameters_off_the_catchment_grid( lake = _LakeStub(coello_loaded.meteo.time_steps) with pytest.raises(ValueError, match="as many rows as the catchment grid"): - Run.RunFW1withLake(coello_loaded, lake) + Run.run_maxbas_with_lake(coello_loaded, lake) - assert "FW1Withlake" not in spied_wrapper, ( + assert "run_maxbas_with_lake" not in spied_wrapper, ( "the wrapper must not run on mis-shaped parameters" ) diff --git a/tests/rrm/catchment/test_save_results_distributed.py b/tests/rrm/catchment/test_save_results_distributed.py index 1df6e380..f14b2947 100644 --- a/tests/rrm/catchment/test_save_results_distributed.py +++ b/tests/rrm/catchment/test_save_results_distributed.py @@ -51,7 +51,7 @@ def coello_run( coello.flow_network = FlowNetwork.from_rasters(coello_acc_path) coello.read_parameters(coello_dist_parameters_maxbas, False, maxbas=True) coello.read_lumped_model(HBVLumped, coello_cat_area, coello_initial_cond) - Run.runFW1(coello) + Run.run_maxbas(coello) return coello @@ -122,7 +122,7 @@ def test_save_results_distributed_values_match_the_model_array( written = sorted(out.glob("*.tif")) start_i = np.where(coello_run.date_index == np.datetime64("2009-01-01"))[0][0] - expected = coello_run.state_variables[:, :, start_i, 0] + expected = coello_run.results.state_variables[:, :, start_i, 0] actual = Dataset.read_file(str(written[0])).read_array(band=0) np.testing.assert_allclose( diff --git a/tests/rrm/catchment/test_wrapper_with_lake.py b/tests/rrm/catchment/test_wrapper_with_lake.py index ee03c05b..93051d46 100644 --- a/tests/rrm/catchment/test_wrapper_with_lake.py +++ b/tests/rrm/catchment/test_wrapper_with_lake.py @@ -1,6 +1,6 @@ """End-to-end tests for the two lake-aware wrapper entry points. -`Wrapper.RRMWithlake` and `Wrapper.FW1Withlake` run the lake as a lumped inflow, add its routed +`Wrapper.run_muskingum_with_lake` and `Wrapper.run_maxbas_with_lake` run the lake as a lumped inflow, add its routed discharge to the outflow cell, and then route the sub-catchment — Muskingum in the first case, triangular (MAXBAS) in the second. Neither had test coverage, so the sizes they read off `MeteoInputs` and `FlowNetwork` and the `_maxbas_routed` flag they leave behind were unpinned. @@ -79,7 +79,7 @@ def _build_coello( area: Catchment area in km2. initial_cond: Initial HBV state. maxbas: Whether `parameters` carries the triangular-routing parameter. The two sets - differ in their last band -- MAXBAS against the Muskingum X -- and `DistMaxbas1` + differ in their last band -- MAXBAS against the Muskingum X -- and `route_maxbas` routes with whatever sits there, so a triangular run needs the MAXBAS set. Returns: @@ -161,7 +161,7 @@ def coello_no_lake( """Provide a second, identical catchment to run without a lake as the control. Returns: - Catchment: Same inputs as `coello_with_lake_inputs`, to be routed by `RRMModel`. + Catchment: Same inputs as `coello_with_lake_inputs`, to be routed by `run_muskingum`. """ return _build_coello( coello_start_date, @@ -192,7 +192,7 @@ def coello_with_lake_inputs_maxbas( ) -> Catchment: """Provide the same catchment carrying a MAXBAS parameter set. - `DistMaxbas1` routes each cell with `parameters[..., -1]`, which is MAXBAS in this set and + `route_maxbas` routes each cell with `parameters[..., -1]`, which is MAXBAS in this set and the Muskingum X in the other. Handed the Muskingum set, every cell routed with X = 0.2 -- below the one whole step a triangle needs -- and `triangular_routing_1` returned an all-zero hydrograph without raising, so the triangular tests asserted their shapes and flags against @@ -273,7 +273,7 @@ def _make_lake( class TestRRMWithLake: - """Tests for `Wrapper.RRMWithlake` (Muskingum routing).""" + """Tests for `Wrapper.run_muskingum_with_lake` (Muskingum routing).""" def test_the_lake_raises_discharge_at_the_outflow_cell( self, @@ -285,8 +285,8 @@ def test_the_lake_raises_discharge_at_the_outflow_cell( """Test that the lake's routed outflow actually reaches the cell it drains into. Test scenario: - Injecting the lake into `quz` at the outflow cell is the one thing `RRMWithlake` - does that `RRMModel` does not. Running the same catchment both ways isolates it: + Injecting the lake into `quz` at the outflow cell is the one thing `run_muskingum_with_lake` + does that `run_muskingum` does not. Running the same catchment both ways isolates it: everything else is identical, so the whole difference is the lake. The increase is *not* `QlakeR` — the lake series is routed a second time, through @@ -296,12 +296,12 @@ def test_the_lake_raises_discharge_at_the_outflow_cell( model = coello_with_lake_inputs lake = _make_lake(model, coello_start_date, coello_end_date, seed=7) - Wrapper.RRMModel(coello_no_lake) - Wrapper.RRMWithlake(model, lake) + Wrapper.run_muskingum(coello_no_lake) + Wrapper.run_muskingum_with_lake(model, lake) row, col = OUTFLOW_CELL - without = coello_no_lake.quz_routed[row, col, :] - with_lake = model.quz_routed[row, col, :] + without = coello_no_lake.results.quz_routed[row, col, :] + with_lake = model.results.quz_routed[row, col, :] assert np.isfinite(with_lake).all(), ( "the outflow cell must stay finite -- an all-zero or all-NaN series here means " @@ -346,15 +346,15 @@ def test_the_routed_lake_series_covers_every_simulation_step( model = coello_with_lake_inputs lake = _make_lake(model, coello_start_date, coello_end_date, seed=11) - Wrapper.RRMWithlake(model, lake) + Wrapper.run_muskingum_with_lake(model, lake) steps = model.meteo.simulation_steps rows, cols = model.flow_network.rows, model.flow_network.cols assert len(lake.QlakeR) == steps, ( f"the routed lake series must be {steps} long, got {len(lake.QlakeR)}" ) - assert model.Qtot.shape == (rows, cols, steps), ( - f"Expected Qtot {(rows, cols, steps)}, got {model.Qtot.shape}" + assert model.results.q_total.shape == (rows, cols, steps), ( + f"Expected q_total {(rows, cols, steps)}, got {model.results.q_total.shape}" ) assert np.isfinite(lake.QlakeR).all(), ( "the routed lake series must be finite; NaN means the outflow cell's Muskingum " @@ -371,7 +371,7 @@ def test_a_muskingum_lake_run_clears_a_flag_a_triangular_run_set( """Test the real handshake: a triangular run sets the flag, a Muskingum run clears it. Test scenario: - The flag tells `extract_discharge` that a cell of `Qtot` is a contribution rather + The flag tells `extract_discharge` that a cell of `q_total` is a contribution rather than a discharge. Setting it by hand would test the assignment against itself, so drive both paths on the same model in the order that makes the flag matter. The two read different parameter layouts -- Muskingum takes bands 10 and 11, the @@ -381,21 +381,21 @@ def test_a_muskingum_lake_run_clears_a_flag_a_triangular_run_set( model = coello_with_lake_inputs_maxbas lake = _make_lake(model, coello_start_date, coello_end_date, seed=13) - Wrapper.FW1Withlake(model, lake) - assert model._maxbas_routed is True, ( + Wrapper.run_maxbas_with_lake(model, lake) + assert not model.results.outlet_shortcut_valid, ( "the triangular path must mark the model before this test means anything" ) model.read_parameters(coello_dist_parameters_muskingum, False) - Wrapper.RRMWithlake(model, lake) + Wrapper.run_muskingum_with_lake(model, lake) - assert model._maxbas_routed is False, ( + assert model.results.outlet_shortcut_valid, ( "a Muskingum lake run makes the outlet-cell shortcut valid again" ) class TestFW1WithLake: - """Tests for `Wrapper.FW1Withlake` (triangular/MAXBAS routing).""" + """Tests for `Wrapper.run_maxbas_with_lake` (triangular/MAXBAS routing).""" def test_fills_the_distributed_output_fields( self, @@ -403,24 +403,24 @@ def test_fills_the_distributed_output_fields( coello_start_date: str, coello_end_date: str, ): - """Test that the triangular lake path populates `Qtot`, `quz_routed`, `qlz_translated`. + """Test that the triangular lake path populates `q_total`, `quz_routed`, `qlz_translated`. Test scenario: `save_results` and `plot_distributed_results` read those three fields, and only the Muskingum path used to set them. The fixture is function-scoped so the model - has been through `FW1Withlake` and nothing else -- with a shared instance an + has been through `run_maxbas_with_lake` and nothing else -- with a shared instance an earlier Muskingum run would have filled them and deleting the fix left this green. """ model = coello_with_lake_inputs_maxbas lake = _make_lake(model, coello_start_date, coello_end_date, seed=17) - assert model.Qtot is None, "the fixture must arrive with no run behind it" + assert model.results is None, "the fixture must arrive with no run behind it" - Wrapper.FW1Withlake(model, lake) + Wrapper.run_maxbas_with_lake(model, lake) rows, cols = model.flow_network.rows, model.flow_network.cols steps = model.meteo.simulation_steps - for name in ("Qtot", "quz_routed", "qlz_translated"): - field = getattr(model, name) + for name in ("q_total", "quz_routed", "qlz_translated"): + field = getattr(model.results, name) assert field is not None, f"{name} must be set by the triangular path" assert field.shape == (rows, cols, steps), ( f"{name} should be {(rows, cols, steps)}, got {field.shape}" @@ -443,21 +443,22 @@ def test_the_outlet_series_carries_the_lake_and_drops_the_extra_slot( model = coello_with_lake_inputs_maxbas lake = _make_lake(model, coello_start_date, coello_end_date, seed=19) - Wrapper.FW1Withlake(model, lake) + Wrapper.run_maxbas_with_lake(model, lake) expected_len = model.meteo.simulation_steps - 1 - assert len(model.qout) == expected_len, ( + assert len(model.results.qout) == expected_len, ( f"qout should drop the trailing slot: expected {expected_len}, " - f"got {len(model.qout)}" + f"got {len(model.results.qout)}" ) subcatchment = np.array( [ - np.nansum(model.qlz[:, :, i]) + np.nansum(model.quz[:, :, i]) + np.nansum(model.results.qlz[:, :, i]) + + np.nansum(model.results.quz[:, :, i]) for i in range(model.meteo.simulation_steps) ] )[:-1] np.testing.assert_allclose( - model.qout, + model.results.qout, subcatchment + lake.QlakeR[:-1], rtol=1e-6, err_msg="qout must be the trimmed sub-catchment sum plus the trimmed lake series", @@ -473,16 +474,16 @@ def test_marks_the_model_as_maxbas_routed( Test scenario: Triangular routing sends every cell straight to the outlet, so reading a gauge - cell of `Qtot` under-reports. The flag is what makes `extract_discharge` refuse + cell of `q_total` under-reports. The flag is what makes `extract_discharge` refuse rather than return the wrong hydrograph. """ model = coello_with_lake_inputs_maxbas lake = _make_lake(model, coello_start_date, coello_end_date, seed=23) - assert model._maxbas_routed is False, "a fresh model must start unflagged" + assert model.results is None, "a fresh model must arrive with no results" - Wrapper.FW1Withlake(model, lake) + Wrapper.run_maxbas_with_lake(model, lake) - assert model._maxbas_routed is True, ( + assert not model.results.outlet_shortcut_valid, ( "a triangular lake run must mark the model as MAXBAS-routed" ) @@ -496,7 +497,7 @@ def test_the_entry_point_runs_with_a_record_of_the_documented_length( coello_start_date: str, coello_end_date: str, ): - """Test that `Run.runHAPIwithLake` validates and then completes. + """Test that `Run.run_distributed_with_lake` validates and then completes. Test scenario: The entry point asserts the lake record covers exactly `meteo.time_steps` and @@ -507,9 +508,11 @@ def test_the_entry_point_runs_with_a_record_of_the_documented_length( model = coello_with_lake_inputs lake = _make_lake(model, coello_start_date, coello_end_date, seed=29) - Run.runHAPIwithLake(model, lake) + Run.run_distributed_with_lake(model, lake) assert lake.MeteoData.shape[0] == model.meteo.time_steps, ( "the record the entry point accepts must be the one the wrapper can run" ) - assert model.Qtot is not None, "the run must leave a routed discharge field" + assert model.results.q_total is not None, ( + "the run must leave a routed discharge field" + ) diff --git a/tests/run/distributed_mode_run.py b/tests/run/distributed_mode_run.py index a8879667..9ccb02a8 100644 --- a/tests/run/distributed_mode_run.py +++ b/tests/run/distributed_mode_run.py @@ -57,7 +57,7 @@ 6-qlz_translated: [numpy attribute] 3D array of the lower zone discharge translated at each time step """ -Run.RunHapi(Coello) +Run.run_distributed(Coello) # %% calculate performance criteria Coello.extract_discharge(factor=Coello.GaugesTable["area ratio"].tolist()) diff --git a/tests/run/lumped_run.py b/tests/run/lumped_run.py index 606c66e1..a09835bb 100644 --- a/tests/run/lumped_run.py +++ b/tests/run/lumped_run.py @@ -38,7 +38,7 @@ routing_fn = Routing.muskingum_v Route = 1 ### run the model -Run.runLumped(Coello, Route, routing_fn) +Run.run_lumped(Coello, Route, routing_fn) # %% calculate performance criteria scores = dict() diff --git a/tests/sensitivity_analysis.py b/tests/sensitivity_analysis.py index ceb0842d..cc2ac3c6 100644 --- a/tests/sensitivity_analysis.py +++ b/tests/sensitivity_analysis.py @@ -53,7 +53,7 @@ routing_fn = Routing.muskingum # %% ### run the model -Run.runLumped(Coello, Route, routing_fn) +Run.run_lumped(Coello, Route, routing_fn) # %% scores = dict() @@ -87,7 +87,7 @@ the following defined function contains two inner functions that calculate discharge for the lumped HBV model and the RMSE of the calculated discharge. - the first function "Run.runLumped" takes some arguments we need to pass through + the first function "Run.run_lumped" takes some arguments we need to pass through the one_at_a_time method [ConceptualModel,data,p2,init_st,snow,Routing, routing_fn] with the same order in the defined function "wrapper" @@ -113,7 +113,7 @@ def WrapperType1(Randpar, Route, routing_fn, Qobs): Coello.parameters = Randpar - Run.runLumped(Coello, Route, routing_fn) + Run.run_lumped(Coello, Route, routing_fn) rmse = metrics.rmse(Qobs, Coello.Qsim["q"]) return rmse @@ -122,7 +122,7 @@ def WrapperType1(Randpar, Route, routing_fn, Qobs): def WrapperType2(Randpar, Route, routing_fn, Qobs): Coello.parameters = Randpar - Run.runLumped(Coello, Route, routing_fn) + Run.run_lumped(Coello, Route, routing_fn) rmse = metrics.rmse(Qobs, Coello.Qsim["q"]) return rmse, Coello.Qsim["q"] From 9feb095a0994fbef7ec10c35870f2687bca8bae9 Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Tue, 1 Sep 2026 23:22:13 +0200 Subject: [PATCH 06/54] refactor(period)!: make the simulation span one object instead of six fields `start`, `end`, `temporal_resolution`, `date_index`, `dt` and `conversion_factor` described one thing. Three of them were inputs; the other three were computed from those in the constructor and then stored alongside, as if independent. Storing a derivation is how they drift: reassigning `end` left `date_index` describing the old span with nothing to notice. It is also why the same `pd.date_range` branch appeared four times across the package. `SimulationPeriod` holds the three inputs and derives the rest on read, so they cannot disagree, and it is frozen so a run covers the span it was built for. It validates the resolution and rejects a backwards span, which previously produced an empty `date_index` that failed much later as a zero-length driver mismatch naming neither date. This is not a typing exercise. The point is that a run should need a handful of coherent things from the model rather than a list of loose fields, and each of them should be constructed-or-absent rather than nullable -- the shape `MeteoInputs` and `FlowNetwork` already established here. The run layer's read surface drops from 22 named attributes to 17, and the four `date_range` copies become one derivation. `Lake` keeps its own `start` / `end` / `Index`: it is a separate object with its own record, not a catchment. BREAKING CHANGE: `Catchment.start`, `.end`, `.date_index`, `.dt`, `.conversion_factor` and `.temporal_resolution` are gone. Read them off `model.period` -- `model.period.date_index`, `model.period.conversion_factor`. `Catchment(...)` still takes the same constructor arguments. --- .../coello-distributed-model-run-maxbas.py | 4 +- .../coello-distributed-model-run-netcdf.py | 16 +- .../run/coello-lumped-model-run-maxbas.py | 4 +- .../coello/run/coello-lumped-model-run.py | 4 +- pyproject.toml | 3 +- src/hapi/calibration.py | 12 +- src/hapi/catchment.py | 67 +++---- src/hapi/period.py | 185 ++++++++++++++++++ src/hapi/protocols.py | 19 +- src/hapi/rrm/distrrm.py | 4 +- src/hapi/run.py | 9 +- src/hapi/wrapper.py | 16 +- tests/rrm/calibration/test_rrm_calibration.py | 2 +- tests/rrm/catchment/test_config.py | 26 +-- .../catchment/test_e2e_coello_from_netcdf.py | 18 +- .../test_extract_discharge_distributed.py | 2 +- tests/rrm/catchment/test_fw1_output_fields.py | 4 +- tests/rrm/catchment/test_meteo_inputs.py | 2 +- .../test_read_parameters_validation.py | 25 +-- tests/rrm/catchment/test_rrm_catchment.py | 6 +- .../test_save_results_distributed.py | 4 +- tests/rrm/catchment/test_wrapper_with_lake.py | 2 +- 22 files changed, 309 insertions(+), 125 deletions(-) create mode 100644 src/hapi/period.py diff --git a/examples/hydrological-model/coello/run/coello-distributed-model-run-maxbas.py b/examples/hydrological-model/coello/run/coello-distributed-model-run-maxbas.py index dd3cc662..04b5cb46 100644 --- a/examples/hydrological-model/coello/run/coello-distributed-model-run-maxbas.py +++ b/examples/hydrological-model/coello/run/coello-distributed-model-run-maxbas.py @@ -56,4 +56,6 @@ print("WB= " + str(round(Coello.metrics.loc["WB", gaugeid], 2))) # %% plot the hydrograph at the outlet gauge (row position, not the gauge id) -Coello.plot_hydrograph(Coello.start, Coello.end, Coello.GaugesTable.index[-1]) +Coello.plot_hydrograph( + Coello.period.start, Coello.period.end, Coello.GaugesTable.index[-1] +) diff --git a/examples/hydrological-model/coello/run/coello-distributed-model-run-netcdf.py b/examples/hydrological-model/coello/run/coello-distributed-model-run-netcdf.py index b582106f..6156af79 100644 --- a/examples/hydrological-model/coello/run/coello-distributed-model-run-netcdf.py +++ b/examples/hydrological-model/coello/run/coello-distributed-model-run-netcdf.py @@ -30,14 +30,16 @@ # %% Check the drivers actually came from the file and cover the model print(f"meteo grid + steps : {Coello.meteo.shape}") -print(f"model steps : {len(Coello.date_index)}") +print(f"model steps : {len(Coello.period.date_index)}") print(f"meteo period : {Coello.meteo.time[0]} -> {Coello.meteo.time[-1]}") -print(f"model period : {Coello.date_index[0]} -> {Coello.date_index[-1]}") -if Coello.meteo.time_steps != len(Coello.date_index): +print( + f"model period : {Coello.period.date_index[0]} -> {Coello.period.date_index[-1]}" +) +if Coello.meteo.time_steps != len(Coello.period.date_index): raise ValueError("the drivers must hold exactly as many steps as the model spans") -if Coello.meteo.time[0] != Coello.date_index[0]: +if Coello.meteo.time[0] != Coello.period.date_index[0]: raise ValueError("the drivers must start where the model does") -if Coello.meteo.time[-1] != Coello.date_index[-1]: +if Coello.meteo.time[-1] != Coello.period.date_index[-1]: raise ValueError("the drivers must end where the model does") # %% Run the model @@ -97,4 +99,6 @@ print(f"rasters written to : {save_to}") # %% Plot the hydrograph at the outlet gauge (row position, not the gauge id) -Coello.plot_hydrograph(Coello.start, Coello.end, Coello.GaugesTable.index[-1]) +Coello.plot_hydrograph( + Coello.period.start, Coello.period.end, Coello.GaugesTable.index[-1] +) diff --git a/examples/hydrological-model/coello/run/coello-lumped-model-run-maxbas.py b/examples/hydrological-model/coello/run/coello-lumped-model-run-maxbas.py index 1d9be4a3..5f959441 100644 --- a/examples/hydrological-model/coello/run/coello-lumped-model-run-maxbas.py +++ b/examples/hydrological-model/coello/run/coello-lumped-model-run-maxbas.py @@ -54,7 +54,9 @@ # %% Plot Hydrograph gaugei = 0 -fig, ax = Coello.plot_hydrograph(Coello.start, Coello.end, gaugei, title="Lumped Model") +fig, ax = Coello.plot_hydrograph( + Coello.period.start, Coello.period.end, gaugei, title="Lumped Model" +) # %% Save Results SaveTo = Coello.config.outputs.results_dir diff --git a/examples/hydrological-model/coello/run/coello-lumped-model-run.py b/examples/hydrological-model/coello/run/coello-lumped-model-run.py index b3ef8efc..ea822730 100644 --- a/examples/hydrological-model/coello/run/coello-lumped-model-run.py +++ b/examples/hydrological-model/coello/run/coello-lumped-model-run.py @@ -54,7 +54,9 @@ # %% Plot Hydrograph gaugei = 0 -fig, ax = Coello.plot_hydrograph(Coello.start, Coello.end, gaugei, title="Lumped Model") +fig, ax = Coello.plot_hydrograph( + Coello.period.start, Coello.period.end, gaugei, title="Lumped Model" +) # %% Save Results SaveTo = Coello.config.outputs.results_dir diff --git a/pyproject.toml b/pyproject.toml index 28733b17..9adfab27 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -261,7 +261,8 @@ description = "Run all test suite" cmd = [ "pytest", "--doctest-modules", "-p", "no:cacheprovider", "--no-cov", "src/hapi/config.py", "src/hapi/catchment.py", - "src/hapi/inputs.py", "src/hapi/results.py", "src/hapi/routing.py", + "src/hapi/inputs.py", "src/hapi/period.py", "src/hapi/results.py", + "src/hapi/routing.py", "src/hapi/run.py", ] description = "Run the doctests of the modules whose examples are executable" diff --git a/src/hapi/calibration.py b/src/hapi/calibration.py index 987abc0f..cbb19578 100644 --- a/src/hapi/calibration.py +++ b/src/hapi/calibration.py @@ -291,7 +291,7 @@ def run_calibration( # The three cubes already agree with each other (checked when MeteoInputs was # built); this is the other half -- that they cover the model's grid. self.meteo.validate_against( - self.flow_network.rows, self.flow_network.cols, self.date_index + self.flow_network.rows, self.flow_network.cols, self.period.date_index ) # basic inputs @@ -330,8 +330,8 @@ def opt_fun(par): for i in range(len(f)): k = par[f[i]] x = par[f[i] + 1] - g.append(2 * k * x / self.dt) - g.append((2 * k * (1 - x)) / self.dt) + g.append(2 * k * x / self.period.dt) + g.append((2 * k * (1 - x)) / self.period.dt) except TypeError as e: # the objective function received fewer inputs than it needs @@ -437,7 +437,7 @@ def calibrate_maxbas( # The three cubes already agree with each other (checked when MeteoInputs was # built); this is the other half -- that they cover the model's grid. self.meteo.validate_against( - self.flow_network.rows, self.flow_network.cols, self.date_index + self.flow_network.rows, self.flow_network.cols, self.period.date_index ) # basic inputs @@ -606,8 +606,8 @@ def opt_fun(par): self.QGauges[self.QGauges.columns[-1]], self.Qsim, *self.OFArgs ) g = [ - 2 * par[-2] * par[-1] / self.dt, - (2 * par[-2] * (1 - par[-1])) / self.dt, + 2 * par[-2] * par[-1] / self.period.dt, + (2 * par[-2] * (1 - par[-1])) / self.period.dt, ] except TypeError as e: # the objective function received fewer inputs than it needs diff --git a/src/hapi/catchment.py b/src/hapi/catchment.py index 7a3702fc..714e8e89 100644 --- a/src/hapi/catchment.py +++ b/src/hapi/catchment.py @@ -43,6 +43,7 @@ _warn_if_no_sentinel, read_rasters, ) +from hapi.period import SimulationPeriod from hapi.results import SimulationResults from hapi.rrm.hbv import HBV from hapi.rrm.hbv_bergestrom92 import HBVBergestrom92 @@ -263,15 +264,9 @@ def __init__( "Kinematic". """ self.name = name - self.start = dt.datetime.strptime(start_data, fmt) - self.end = dt.datetime.strptime(end, fmt) - # All three of the mode arguments are lower-cased below, so a non-string reaches - # `.lower()` and raises an `AttributeError` naming neither the argument nor the class. - # Checked together, once, rather than three times over. for argument, value in ( ("spatial_resolution", spatial_resolution), - ("temporal_resolution", temporal_resolution), ("routing_method", routing_method), ): if not isinstance(value, str): @@ -285,22 +280,14 @@ def __init__( ) self.spatial_resolution = spatial_resolution.lower() - if temporal_resolution.lower() not in ["daily", "hourly"]: - raise ValueError("available temporal resolutions are 'daily' and 'hourly'") - self.temporal_resolution = temporal_resolution.lower() - # assuming the default dt is 1 day - # Only the two resolutions the check above admits: an `else` here would be - # unreachable, and the one that used to sit here set a conversion factor but no - # `date_index`, which reads as support for sub-daily steps that does not exist. - # Adding one (q mm, area km2: 1/(3.6*f)) means widening the check above too. - if self.temporal_resolution == "daily": - self.dt = 1 # 24 - self.conversion_factor = CONVERSION_FACTOR * 1 - self.date_index = pd.date_range(self.start, self.end, freq="D") - else: - self.dt = 1 # 24 - self.conversion_factor = CONVERSION_FACTOR * 1 / 24 - self.date_index = pd.date_range(self.start, self.end, freq="h") + #: The span this model runs over. One object rather than six attributes: `start`, + #: `end` and `temporal_resolution` are the inputs, and `date_index`, `dt` and + #: `conversion_factor` are derived from them on read, so they cannot describe a + #: different span from the one the model is set to. It validates the resolution and + #: rejects a backwards span. + self.period = SimulationPeriod.parse( + start_data, end, fmt=fmt, temporal_resolution=temporal_resolution + ) # Canonicalised like the two resolutions above, and for a sharper reason: # `distrrm.route_muskingum` tests `routing_method != "Muskingum"` case-sensitively, and @@ -399,7 +386,7 @@ def from_yaml(cls, path: str | Path) -> Self: 'Coello' >>> model.spatial_resolution 'lumped' - >>> len(model.date_index) + >>> len(model.period.date_index) 1095 ``` @@ -970,10 +957,10 @@ def read_discharge_gauges( ValueError: If the gauge table has not been read yet (distributed mode). """ - if self.temporal_resolution.lower() == "daily": - ind = pd.date_range(self.start, self.end, freq="D") + if self.period.temporal_resolution.lower() == "daily": + ind = pd.date_range(self.period.start, self.period.end, freq="D") else: - ind = pd.date_range(self.start, self.end, freq="h") + ind = pd.date_range(self.period.start, self.period.end, freq="h") if self.spatial_resolution.lower() == "distributed": self._read_one_discharge_file_per_gauge( @@ -1047,7 +1034,9 @@ def _read_one_discharge_file_per_gauge( ) f.index = [dt.datetime.strptime(i, fmt) for i in f.index.tolist()] - self.QGauges[labels[i]] = f.loc[self.start : self.end, f.columns[-1]] + self.QGauges[labels[i]] = f.loc[ + self.period.start : self.period.end, f.columns[-1] + ] def _read_the_single_discharge_file( self, path: str, index: pd.DatetimeIndex, delimiter: str, fmt: str @@ -1072,7 +1061,9 @@ def _read_the_single_discharge_file( self.QGauges = pd.DataFrame(index=index) f = pd.read_csv(path, header=0, index_col=0, delimiter=delimiter) f.index = [dt.datetime.strptime(i, fmt) for i in f.index.tolist()] - self.QGauges[f.columns[0]] = f.loc[self.start : self.end, f.columns[0]] + self.QGauges[f.columns[0]] = f.loc[ + self.period.start : self.period.end, f.columns[0] + ] def read_parameters_bound( self, @@ -1153,7 +1144,7 @@ def extract_discharge(self, calculate_metrics=True, factor=None): if self.results.outlet_shortcut_valid: self.Qsim = pd.DataFrame( - index=self.date_index, columns=self.QGauges.columns + index=self.period.date_index, columns=self.QGauges.columns ) if calculate_metrics: index = ["RMSE", "NSE", "NSEhf", "KGE", "WB", "Pearson-CC", "R2"] @@ -1210,7 +1201,7 @@ def extract_discharge(self, calculate_metrics=True, factor=None): else: # MAXBAS: a cell of `q_total` is a contribution, so the hydrograph is the # basin-wide sum the run already put in `qout`. - self.Qsim = pd.DataFrame(index=self.date_index) + self.Qsim = pd.DataFrame(index=self.period.date_index) gauge_id = self.GaugesTable.loc[self.GaugesTable.index[-1], "id"] q_sim = np.reshape(self.results.qout, self.meteo.time_steps) self.Qsim.loc[:, gauge_id] = q_sim @@ -1432,8 +1423,8 @@ def plot_distributed_results( start = dt.datetime.strptime(start, fmt) end = dt.datetime.strptime(end, fmt) - start_i = np.nonzero(self.date_index == start)[0][0] - end_i = np.nonzero(self.date_index == end)[0][0] + start_i = np.nonzero(self.period.date_index == start)[0][0] + end_i = np.nonzero(self.period.date_index == end)[0][0] if option == 1: arr = self.results.q_total[:, :, start_i:end_i] @@ -1476,7 +1467,7 @@ def plot_distributed_results( arr = arr.copy() arr[np.isnan(self.flow_network.flow_acc_arr), :] = np.nan - time = self.date_index[start_i:end_i] + time = self.period.date_index[start_i:end_i] if gauges: # animate expects a 3-column array: [value to display, cell row, cell column]. @@ -1575,17 +1566,17 @@ def save_results( ) if start == "": - start = self.date_index[0] + start = self.period.date_index[0] elif isinstance(start, str): start = dt.datetime.strptime(start, fmt) if end == "": - end = self.date_index[-1] + end = self.period.date_index[-1] elif isinstance(end, str): end = dt.datetime.strptime(end, fmt) - start_i = np.nonzero(self.date_index == start)[0][0] - end_i = np.nonzero(self.date_index == end)[0][0] + 1 + start_i = np.nonzero(self.period.date_index == start)[0][0] + end_i = np.nonzero(self.period.date_index == end)[0][0] + 1 if self.spatial_resolution == "distributed": if flow_acc_path == "": @@ -1606,7 +1597,7 @@ def save_results( os.makedirs(path, exist_ok=True) names = [ os.path.join(path, f"{prefix}{str(i)[:10]}.tif") - for i in self.date_index[start_i:end_i] + for i in self.period.date_index[start_i:end_i] ] if result == 1: arr = self.results.q_total[:, :, start_i:end_i] diff --git a/src/hapi/period.py b/src/hapi/period.py new file mode 100644 index 00000000..53c00a2f --- /dev/null +++ b/src/hapi/period.py @@ -0,0 +1,185 @@ +"""The span of time a model run covers, and everything the temporal resolution implies. + +Six attributes on :class:`~hapi.catchment.Catchment` used to describe one thing: `start`, +`end` and `temporal_resolution` were given, and `date_index`, `dt` and `conversion_factor` +were derived from them in the constructor and then stored beside them as if they were +independent. Storing a derivation is how the three drift apart -- reassigning `end` left +`date_index` describing the old span, with nothing to notice -- and it is why the same +`pd.date_range` branch was written out four times across the package. + +:class:`SimulationPeriod` holds the three inputs and derives the rest on read, so they cannot +disagree. It is frozen: a run covers the period it was built for, and a model that needs a +different one gets a new period rather than a mutated one. +""" + +from __future__ import annotations + +import datetime as dt +from dataclasses import dataclass +from typing import Literal + +import pandas as pd + +#: mm over a km2 in a day, expressed as m3/s: (1000 * 24 * 60 * 60) / (1000 ** 2). +CONVERSION_FACTOR = (1000 * 24 * 60 * 60) / (1000**2) + +#: The temporal resolutions the model runs at, mapped to the pandas offset alias each uses. +#: Adding one means deciding its `conversion_factor` too -- see :attr:`SimulationPeriod.freq`. +RESOLUTIONS: dict[str, str] = {"daily": "D", "hourly": "h"} + +TemporalResolution = Literal["daily", "hourly"] + + +@dataclass(frozen=True) +class SimulationPeriod: + """The span a run covers, and the calendar and unit factors it implies. + + Attributes: + start: First step of the simulation. + end: Last step of the simulation. + temporal_resolution: `"daily"` or `"hourly"`, lower-cased on construction. + + Examples: + - The calendar is derived, so it always matches the span: + ```python + >>> from hapi.period import SimulationPeriod + >>> period = SimulationPeriod.parse("2009-01-01", "2009-01-10") + >>> len(period) + 10 + >>> period.date_index[0].strftime("%Y-%m-%d") + '2009-01-01' + + ``` + - An hourly period covers the same span with a different step: + ```python + >>> from hapi.period import SimulationPeriod + >>> hourly = SimulationPeriod.parse( + ... "2009-01-01", "2009-01-02", temporal_resolution="Hourly" + ... ) + >>> len(hourly) + 25 + >>> round(hourly.conversion_factor, 1) + 3.6 + + ``` + - It is frozen, so a derived value can never be left describing a different span: + ```python + >>> from hapi.period import SimulationPeriod + >>> period = SimulationPeriod.parse("2009-01-01", "2009-01-10") + >>> period.end = "2010-01-01" # doctest: +ELLIPSIS + Traceback (most recent call last): + ... + dataclasses.FrozenInstanceError: cannot assign to field 'end' + + ``` + """ + + start: dt.datetime + end: dt.datetime + temporal_resolution: TemporalResolution = "daily" + + def __post_init__(self): + """Normalise the resolution and check the span runs forwards. + + Raises: + TypeError: `temporal_resolution` is not a string. + ValueError: The resolution is not one of :data:`RESOLUTIONS`, or `end` is before + `start`. + """ + if not isinstance(self.temporal_resolution, str): + raise TypeError( + f"temporal_resolution must be a string, got " + f"{type(self.temporal_resolution).__name__}" + ) + resolution = self.temporal_resolution.lower() + if resolution not in RESOLUTIONS: + raise ValueError( + f"available temporal resolutions are {', '.join(map(repr, RESOLUTIONS))}, " + f"got {self.temporal_resolution!r}" + ) + object.__setattr__(self, "temporal_resolution", resolution) + + # A backwards span produces an empty `date_index`, which then fails much later as a + # zero-length driver mismatch that names neither date. + if self.end < self.start: + raise ValueError( + f"the simulation ends before it starts: {self.start:%Y-%m-%d} to " + f"{self.end:%Y-%m-%d}" + ) + + @classmethod + def parse( + cls, + start: str, + end: str, + fmt: str = "%Y-%m-%d", + temporal_resolution: str = "Daily", + ) -> SimulationPeriod: + """Build a period from the string dates a configuration or a script supplies. + + Args: + start: Start date. + end: End date. + fmt: `strptime` format both dates are read with. + temporal_resolution: `"Daily"` or `"Hourly"`, matched case-insensitively. + + Returns: + SimulationPeriod: The parsed period. + + Raises: + ValueError: A date does not match `fmt`, or the span runs backwards. + + Examples: + ```python + >>> from hapi.period import SimulationPeriod + >>> SimulationPeriod.parse("01/2009/01", "10/2009/01", fmt="%d/%Y/%m").days + 10 + + ``` + """ + return cls( + dt.datetime.strptime(start, fmt), + dt.datetime.strptime(end, fmt), + temporal_resolution, # type: ignore[arg-type] + ) + + @property + def freq(self) -> str: + """str: The pandas offset alias for this resolution.""" + return RESOLUTIONS[self.temporal_resolution] + + @property + def date_index(self) -> pd.DatetimeIndex: + """pandas.DatetimeIndex: One entry per step, from :attr:`start` to :attr:`end`. + + Derived rather than stored: this is the value that used to be computed in the + constructor and could then outlive a change to the span it described. + """ + return pd.date_range(self.start, self.end, freq=self.freq) + + @property + def days(self) -> int: + """int: Number of steps the period covers.""" + return len(self.date_index) + + @property + def conversion_factor(self) -> float: + """float: Depth-to-discharge factor -- mm over the catchment to m3/s at this step.""" + return CONVERSION_FACTOR if self.temporal_resolution == "daily" else ( + CONVERSION_FACTOR / 24 + ) + + @property + def dt(self) -> float: + """float: The routing time-step factor. + + One for both resolutions today. It is a property rather than a stored `1` so the + Muskingum routing has a single place to read it from; whether an hourly run should + route with a different value is an open question, recorded in the planning notes + rather than silently answered here. + """ + return 1.0 + + def __len__(self) -> int: + """int: Number of steps, so `len(period)` reads as the span.""" + return self.days diff --git a/src/hapi/protocols.py b/src/hapi/protocols.py index 7491cc06..2dcd4ea3 100644 --- a/src/hapi/protocols.py +++ b/src/hapi/protocols.py @@ -17,13 +17,12 @@ from __future__ import annotations -import datetime as dt from typing import TYPE_CHECKING, Any, Protocol import numpy as np -import pandas as pd from hapi.inputs import FlowNetwork, MeteoInputs +from hapi.period import SimulationPeriod from hapi.results import SimulationResults if TYPE_CHECKING: @@ -44,19 +43,19 @@ class ConceptualModelInputs(Protocol): q_init: Initial discharge in m3/s, or None to let the model choose. snow: 1 to run the snow routine, 0 otherwise. area: Catchment area in km2. - conversion_factor: Depth-to-discharge factor for the temporal resolution. - dt: Time-step factor used by the Muskingum routing. + period: The span the run covers, and the calendar, `dt` and `conversion_factor` + it implies. One object rather than six loose fields, so a derived value can + never describe a different span from the one the model is set to. results: Where the run writes its output. None before the first run. """ + period: SimulationPeriod parameters: np.ndarray | list lumped_model: BaseConceptualModel initial_cond: list q_init: float | None snow: int area: float | int - conversion_factor: float - dt: float results: SimulationResults | None @@ -66,7 +65,6 @@ class DistributedModel(ConceptualModelInputs, Protocol): Attributes: meteo: The three driver cubes and the calendar they cover. flow_network: The routing network and the grid it defines. - date_index: The model's own calendar, which the drivers are checked against. routing_method: Canonicalised routing method. `route_muskingum` compares this against `"Muskingum"` exactly to decide whether a cell is routed or skipped. bankfull_depth: Read only when `routing_method` is not `"Muskingum"`; None otherwise. @@ -74,7 +72,6 @@ class DistributedModel(ConceptualModelInputs, Protocol): meteo: MeteoInputs flow_network: FlowNetwork - date_index: pd.DatetimeIndex routing_method: str bankfull_depth: np.ndarray | None @@ -86,17 +83,11 @@ class LumpedModelInputs(ConceptualModelInputs, Protocol): data: `(time, 4)` array of precipitation, ET, temperature and the long-term average. maxbas: Whether the parameter vector carries a MAXBAS value, which changes how the routing function is called. - temporal_resolution: `"daily"` or `"hourly"`. - start: First step of the simulation period. - end: Last step of the simulation period. Qsim: Where the routed hydrograph lands. """ data: np.ndarray maxbas: bool - temporal_resolution: str - start: dt.datetime - end: dt.datetime Qsim: Any diff --git a/src/hapi/rrm/distrrm.py b/src/hapi/rrm/distrrm.py index b9684a77..4f760dc5 100644 --- a/src/hapi/rrm/distrrm.py +++ b/src/hapi/rrm/distrrm.py @@ -112,7 +112,7 @@ def run_lumped_model(Model) -> SimulationResults: ) area_coef = Model.area / Model.flow_network.px_tot_area - factor = Model.flow_network.px_area * area_coef / Model.conversion_factor + factor = Model.flow_network.px_area * area_coef / Model.period.conversion_factor # convert quz and qlz from mm/time step to m3/sec # Timef*3.6 results.quz = results.quz * factor results.qlz = results.qlz * factor @@ -224,7 +224,7 @@ def route_muskingum(Model): results.quz_routed[x_ind, y_ind, 0], Model.parameters[x_ind, y_ind, 10], Model.parameters[x_ind, y_ind, 11], - Model.dt, + Model.period.dt, ) qlzi = qlzi + results.qlz_translated[x_ind, y_ind, :] diff --git a/src/hapi/run.py b/src/hapi/run.py index 21f302e1..f1c9bfbc 100644 --- a/src/hapi/run.py +++ b/src/hapi/run.py @@ -118,7 +118,7 @@ def _validate_distributed(model: DistributedModel, check_flow_direction: bool) - # The three cubes already agree with each other (checked when MeteoInputs was # built); this is the other half -- that they cover the model's grid. model.meteo.validate_against( - model.flow_network.rows, model.flow_network.cols, model.date_index + model.flow_network.rows, model.flow_network.cols, model.period.date_index ) _check_parameters_cover_grid(model) @@ -382,10 +382,9 @@ def run_lumped( """ if routing_fn is None and Route != 0: raise ValueError("routing_fn must be a callable when Route != 0") - if model.temporal_resolution.lower() == "daily": - ind = pd.date_range(model.start, model.end, freq="D") - else: - ind = pd.date_range(model.start, model.end, freq="h") + # The calendar belongs to the period, which derives it from the span and the + # resolution -- this branch used to be written out here for the fourth time. + ind = model.period.date_index Qsim = pd.DataFrame(index=ind) diff --git a/src/hapi/wrapper.py b/src/hapi/wrapper.py index 44195f5e..719d85fa 100644 --- a/src/hapi/wrapper.py +++ b/src/hapi/wrapper.py @@ -136,7 +136,7 @@ def run_muskingum_with_lake( t, et, Lake.Parameters, - [Model.conversion_factor, Lake.CatArea, Lake.LakeArea], + [Model.period.conversion_factor, Lake.CatArea, Lake.LakeArea], Lake.StageDischargeCurve, 0, init_st=Lake.InitialCond, @@ -150,7 +150,7 @@ def run_muskingum_with_lake( Lake.Qlake[0], Lake.Parameters[11], Lake.Parameters[12], - Model.conversion_factor, + Model.period.conversion_factor, ) # subcatchment @@ -162,7 +162,7 @@ def run_muskingum_with_lake( Lake.QlakeR[0], Model.parameters[Lake.OutflowCell[0], Lake.OutflowCell[1], 10], Model.parameters[Lake.OutflowCell[0], Lake.OutflowCell[1], 11], - Model.conversion_factor, + Model.period.conversion_factor, ) # No padding: `HBVLake.simulate` already prepends the initial-state slot, exactly as @@ -308,7 +308,7 @@ def run_maxbas_with_lake( t, et, Lake.Parameters, - [Model.conversion_factor, Lake.CatArea, Lake.LakeArea], + [Model.period.conversion_factor, Lake.CatArea, Lake.LakeArea], Lake.StageDischargeCurve, 0, init_st=Lake.InitialCond, @@ -323,7 +323,7 @@ def run_maxbas_with_lake( Lake.Qlake[0], Lake.Parameters[11], Lake.Parameters[12], - Model.conversion_factor, + Model.period.conversion_factor, ) # subcatchment @@ -345,7 +345,7 @@ def run_maxbas_with_lake( qout = qlz1 + quz1 - # qout = (qlz1 + quz1) * Model.CatArea / (Model.conversion_factor* 3.6) + # qout = (qlz1 + quz1) * Model.CatArea / (Model.period.conversion_factor* 3.6) # Both series run over `simulation_steps`, and the non-lake FW1 path returns # `qout[:-1]` -- dropping the trailing slot, not the leading initial-state one. The @@ -428,7 +428,7 @@ def run_lumped( ) # q mm , area sq km (1000**2)/1000/f/60/60 = 1/(3.6*f) # if daily tfac=24 if hourly tfac=1 if 15 min tfac=0.25 - factor = Model.area / Model.conversion_factor + factor = Model.area / Model.period.conversion_factor # A lumped run has no spatial routing at all, so the routed fields stay None and # the routing kind says why -- rather than a MAXBAS flag left over from elsewhere. results = SimulationResults( @@ -449,7 +449,7 @@ def run_lumped( Model.Qsim[0], Model.parameters[-2], Model.parameters[-1], - Model.dt, + Model.period.dt, ) return results diff --git a/tests/rrm/calibration/test_rrm_calibration.py b/tests/rrm/calibration/test_rrm_calibration.py index bd4e99f4..8ac25715 100644 --- a/tests/rrm/calibration/test_rrm_calibration.py +++ b/tests/rrm/calibration/test_rrm_calibration.py @@ -98,7 +98,7 @@ def test_create_calibration_instance( ) assert coello.spatial_resolution == "distributed" assert coello.routing_method == "Muskingum" - assert isinstance(coello.start, dt.datetime) + assert isinstance(coello.period.start, dt.datetime) def test_read_objective_fn(self, coello_start_date: str, coello_end_date: str): coello = Calibration( diff --git a/tests/rrm/catchment/test_config.py b/tests/rrm/catchment/test_config.py index 556f02cb..f0da4fae 100644 --- a/tests/rrm/catchment/test_config.py +++ b/tests/rrm/catchment/test_config.py @@ -1345,12 +1345,12 @@ def test_the_drivers_cover_the_model_period(self, distributed_mapping, tmp_path) """ model = Catchment.from_yaml(write_yaml(distributed_mapping, tmp_path)) - assert model.meteo.time_steps == len(model.date_index), ( + assert model.meteo.time_steps == len(model.period.date_index), ( f"drivers hold {model.meteo.time_steps} steps, model spans " - f"{len(model.date_index)}" + f"{len(model.period.date_index)}" ) - assert model.meteo.time[0] == model.date_index[0], ( - f"drivers start at {model.meteo.time[0]}, model at {model.date_index[0]}" + assert model.meteo.time[0] == model.period.date_index[0], ( + f"drivers start at {model.meteo.time[0]}, model at {model.period.date_index[0]}" ) def test_an_explicit_meteo_window_overrides_the_catchment_dates( @@ -1586,13 +1586,13 @@ def test_the_inherited_window_is_parsed_with_the_catchment_format( model = Catchment.from_yaml(write_yaml(distributed_mapping, tmp_path)) - assert model.meteo.time_steps == len(model.date_index), ( + assert model.meteo.time_steps == len(model.period.date_index), ( f"drivers hold {model.meteo.time_steps} steps, model spans " - f"{len(model.date_index)}" + f"{len(model.period.date_index)}" ) - assert model.meteo.time[0] == model.date_index[0], ( + assert model.meteo.time[0] == model.period.date_index[0], ( f"window start {model.meteo.time[0]} does not match the model's " - f"{model.date_index[0]}" + f"{model.period.date_index[0]}" ) def test_a_stated_meteo_window_uses_the_meteo_format( @@ -2001,10 +2001,10 @@ def test_an_unquoted_date_builds_the_same_model_as_a_quoted_one( other.mkdir() unquoted = Catchment.from_yaml(write_yaml(unquoted_mapping, other)) - assert unquoted.start == quoted.start, ( - f"expected {quoted.start}, got {unquoted.start}" + assert unquoted.period.start == quoted.period.start, ( + f"expected {quoted.period.start}, got {unquoted.period.start}" ) - assert len(unquoted.date_index) == len(quoted.date_index), ( + assert len(unquoted.period.date_index) == len(quoted.period.date_index), ( "both spellings should span the same period" ) @@ -2041,9 +2041,9 @@ def test_a_raster_source_configuration_reads_the_three_folders( model = Catchment.from_yaml(write_yaml(distributed_mapping, tmp_path)) - assert model.meteo.time_steps == len(model.date_index), ( + assert model.meteo.time_steps == len(model.period.date_index), ( f"the drivers must span the model period: {model.meteo.time_steps} against " - f"{len(model.date_index)}" + f"{len(model.period.date_index)}" ) def test_a_missing_driver_folder_is_reported_before_anything_is_read( diff --git a/tests/rrm/catchment/test_e2e_coello_from_netcdf.py b/tests/rrm/catchment/test_e2e_coello_from_netcdf.py index 74df504f..f506602a 100644 --- a/tests/rrm/catchment/test_e2e_coello_from_netcdf.py +++ b/tests/rrm/catchment/test_e2e_coello_from_netcdf.py @@ -199,16 +199,16 @@ def test_the_drivers_come_from_the_file_and_cover_the_model( assert model.meteo.shape == (13, 14, 10), ( f"expected a 13x14 grid over 10 steps, got {model.meteo.shape}" ) - assert model.meteo.time_steps == len(model.date_index), ( + assert model.meteo.time_steps == len(model.period.date_index), ( f"the drivers hold {model.meteo.time_steps} steps but the model spans " - f"{len(model.date_index)}" + f"{len(model.period.date_index)}" ) assert model.meteo.time is not None, "the calendar must come out of the file" - assert model.meteo.time[0] == model.date_index[0], ( - f"the drivers start at {model.meteo.time[0]}, the model at {model.date_index[0]}" + assert model.meteo.time[0] == model.period.date_index[0], ( + f"the drivers start at {model.meteo.time[0]}, the model at {model.period.date_index[0]}" ) - assert model.meteo.time[-1] == model.date_index[-1], ( - f"the drivers end at {model.meteo.time[-1]}, the model at {model.date_index[-1]}" + assert model.meteo.time[-1] == model.period.date_index[-1], ( + f"the drivers end at {model.meteo.time[-1]}, the model at {model.period.date_index[-1]}" ) def test_routing_fills_the_distributed_fields(self, muskingum_run: Catchment): @@ -259,7 +259,7 @@ def test_gauge_extraction_and_metrics(self, muskingum_run: Catchment): assert np.isfinite(model.metrics.to_numpy(dtype=float)).all(), ( "every metric must be finite" ) - assert model.Qsim.shape == (len(model.date_index), n_gauges), ( + assert model.Qsim.shape == (len(model.period.date_index), n_gauges), ( f"Qsim shape mismatch: {model.Qsim.shape}" ) assert np.isfinite(model.Qsim.to_numpy(dtype=float)).all(), ( @@ -285,8 +285,8 @@ def test_saved_rasters_carry_the_routed_discharge( written = sorted(out.glob("*.tif")) assert written, "save_results must write at least one raster" - assert len(written) == len(model.date_index), ( - f"expected one raster per step ({len(model.date_index)}), got {len(written)}" + assert len(written) == len(model.period.date_index), ( + f"expected one raster per step ({len(model.period.date_index)}), got {len(written)}" ) first = Dataset.read_file(str(written[0])).read_array() diff --git a/tests/rrm/catchment/test_extract_discharge_distributed.py b/tests/rrm/catchment/test_extract_discharge_distributed.py index 856e17af..ffb50c1e 100644 --- a/tests/rrm/catchment/test_extract_discharge_distributed.py +++ b/tests/rrm/catchment/test_extract_discharge_distributed.py @@ -83,7 +83,7 @@ def test_extract_discharge_distributed_metrics(coello_muskingum_run: Catchment): assert np.isfinite(coello.metrics.to_numpy(dtype=float)).all(), ( "All metric values should be finite" ) - assert coello.Qsim.shape == (len(coello.date_index), n_gauges), ( + assert coello.Qsim.shape == (len(coello.period.date_index), n_gauges), ( f"Qsim shape mismatch: {coello.Qsim.shape}" ) diff --git a/tests/rrm/catchment/test_fw1_output_fields.py b/tests/rrm/catchment/test_fw1_output_fields.py index dbd8f55e..c4fdbb81 100644 --- a/tests/rrm/catchment/test_fw1_output_fields.py +++ b/tests/rrm/catchment/test_fw1_output_fields.py @@ -239,7 +239,9 @@ def test_save_results_distributed_discharge_after_fw1( written = sorted(out.glob("*.tif")) assert len(written) == 4, f"expected one raster per date, got {len(written)}" - start_i = np.where(coello_fw1.date_index == np.datetime64("2009-01-01"))[0][0] + start_i = np.where(coello_fw1.period.date_index == np.datetime64("2009-01-01"))[0][ + 0 + ] np.testing.assert_allclose( Dataset.read_file(str(written[0])).read_array(band=0), coello_fw1.results.q_total[:, :, start_i], diff --git a/tests/rrm/catchment/test_meteo_inputs.py b/tests/rrm/catchment/test_meteo_inputs.py index 1d460fee..a3f6e681 100644 --- a/tests/rrm/catchment/test_meteo_inputs.py +++ b/tests/rrm/catchment/test_meteo_inputs.py @@ -1098,7 +1098,7 @@ def test_metrics(self, coello_muskingum_from_netcdf: Catchment): assert np.isfinite(coello.metrics.to_numpy(dtype=float)).all(), ( "all metric values should be finite" ) - assert coello.Qsim.shape == (len(coello.date_index), n_gauges), ( + assert coello.Qsim.shape == (len(coello.period.date_index), n_gauges), ( f"Qsim shape mismatch: {coello.Qsim.shape}" ) diff --git a/tests/rrm/catchment/test_read_parameters_validation.py b/tests/rrm/catchment/test_read_parameters_validation.py index a2b8e1fd..b9b95a57 100644 --- a/tests/rrm/catchment/test_read_parameters_validation.py +++ b/tests/rrm/catchment/test_read_parameters_validation.py @@ -61,17 +61,17 @@ def test_known_resolutions_build_the_date_index( "coello", "2009-01-01", "2010-01-01", temporal_resolution=resolution ) - assert len(model.date_index) == expected_steps, ( - f"Expected {expected_steps} steps, got {len(model.date_index)}" + assert len(model.period.date_index) == expected_steps, ( + f"Expected {expected_steps} steps, got {len(model.period.date_index)}" ) - assert model.date_index.freqstr.lower() == expected_freq.lower(), ( - f"Expected frequency {expected_freq}, got {model.date_index.freqstr}" + assert model.period.date_index.freqstr.lower() == expected_freq.lower(), ( + f"Expected frequency {expected_freq}, got {model.period.date_index.freqstr}" ) # `dt` is hard-coded to 1 in both branches, so it carries no resolution information; # what distinguishes them is the conversion factor, asserted below. - assert model.dt == 1, f"Expected dt of 1, got {model.dt}" - assert model.temporal_resolution == resolution.lower(), ( - f"the resolution must be stored lowercased, got {model.temporal_resolution}" + assert model.period.dt == 1, f"Expected dt of 1, got {model.period.dt}" + assert model.period.temporal_resolution == resolution.lower(), ( + f"the resolution must be stored lowercased, got {model.period.temporal_resolution}" ) def test_unknown_resolution_is_rejected_at_construction(self): @@ -82,7 +82,7 @@ def test_unknown_resolution_is_rejected_at_construction(self): positionally, so a resolution the constructor cannot build an index for has to fail at construction rather than leave the model half-built. """ - with pytest.raises(ValueError, match="'daily' and 'hourly'"): + with pytest.raises(ValueError, match="temporal resolutions"): Catchment("coello", "2009-01-01", "2009-01-10", temporal_resolution="15min") def test_hourly_resolution_scales_the_conversion_factor(self): @@ -98,9 +98,12 @@ def test_hourly_resolution_scales_the_conversion_factor(self): "coello", "2009-01-01", "2009-01-10", temporal_resolution="Hourly" ) - assert hourly.conversion_factor == pytest.approx( - daily.conversion_factor / 24 - ), f"Expected {daily.conversion_factor / 24}, got {hourly.conversion_factor}" + assert hourly.period.conversion_factor == pytest.approx( + daily.period.conversion_factor / 24 + ), ( + f"Expected {daily.period.conversion_factor / 24}, got " + f"{hourly.period.conversion_factor}" + ) class TestReadParametersDistributed: diff --git a/tests/rrm/catchment/test_rrm_catchment.py b/tests/rrm/catchment/test_rrm_catchment.py index 5e04d731..bd9223d2 100644 --- a/tests/rrm/catchment/test_rrm_catchment.py +++ b/tests/rrm/catchment/test_rrm_catchment.py @@ -15,8 +15,8 @@ def test_create_catchment_instance(coello_rrm_date: list): coello = Catchment("rrm", coello_rrm_date[0], coello_rrm_date[1]) - assert coello.dt == 1 - assert isinstance(coello.date_index, DatetimeIndex) + assert coello.period.dt == 1 + assert isinstance(coello.period.date_index, DatetimeIndex) assert isinstance(coello.routing_method, str) @@ -149,7 +149,7 @@ def test_create_catchment_instance( ) assert coello.spatial_resolution == "distributed" assert coello.routing_method == "Muskingum" - assert isinstance(coello.start, dt.datetime) + assert isinstance(coello.period.start, dt.datetime) def test_read_meteo_inputs( self, diff --git a/tests/rrm/catchment/test_save_results_distributed.py b/tests/rrm/catchment/test_save_results_distributed.py index f14b2947..e025052f 100644 --- a/tests/rrm/catchment/test_save_results_distributed.py +++ b/tests/rrm/catchment/test_save_results_distributed.py @@ -121,7 +121,9 @@ def test_save_results_distributed_values_match_the_model_array( ) written = sorted(out.glob("*.tif")) - start_i = np.where(coello_run.date_index == np.datetime64("2009-01-01"))[0][0] + start_i = np.where(coello_run.period.date_index == np.datetime64("2009-01-01"))[0][ + 0 + ] expected = coello_run.results.state_variables[:, :, start_i, 0] actual = Dataset.read_file(str(written[0])).read_array(band=0) diff --git a/tests/rrm/catchment/test_wrapper_with_lake.py b/tests/rrm/catchment/test_wrapper_with_lake.py index 93051d46..bfd98ca1 100644 --- a/tests/rrm/catchment/test_wrapper_with_lake.py +++ b/tests/rrm/catchment/test_wrapper_with_lake.py @@ -316,7 +316,7 @@ def test_the_lake_raises_discharge_at_the_outflow_cell( lake.QlakeR[0], model.parameters[row, col, 10], model.parameters[row, col, 11], - model.conversion_factor, + model.period.conversion_factor, ) np.testing.assert_allclose( with_lake - without, From 6ac7b62c871e1eab95a902127b8f8310b721d8a9 Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Tue, 1 Sep 2026 23:45:43 +0200 Subject: [PATCH 07/54] refactor(conceptual)!: group the parameter and model fields, and enforce their rule Seven attributes described the conceptual model configured to run -- `parameters`, `snow`, `maxbas`, `lumped_model`, `area`, `initial_cond`, `q_init` -- set by two readers that did not know about each other. The cost was not untidiness. `snow` and `maxbas` *determine how many parameters there must be* (PARAMETER_COUNTS), and that rule was enforced in exactly one place: inside `read_parameters`, on the one assignment a calibration never makes. `calibration.py` replaced the array in six places, three of them inside the optimiser loop -- once per trial vector, thousands of times per run -- and none was checked. A distribution function producing the wrong width reached the per-cell loop and failed there as an index error, far from the call that caused it. Three objects now, one per reader, so no half-built state has to exist: ParameterSet(values, snow, maxbas) <- read_parameters ConceptualModelSetup(model, area, ...) <- read_lumped_model ParameterBounds(lower, upper, snow, maxbas) <- read_parameters_bound A first attempt paired the first two into one spec with a staging step. That was wrong: the readers are independent and either may run first, so pairing them forced exactly the half-built object this is removing, and it broke tests that exercise one reader alone. Splitting on the reader boundary removed the staging entirely. `ParameterSet` is frozen, and the calibration loops go through `with_values`, so every trial vector is checked as it arrives. `ParameterBounds` carries the `(snow, maxbas)` pair too, because a calibration reads no parameter file -- the bounds are where the configuration enters. The optimiser's answer moves to `Calibration.best_parameters`. It was assigned to `parameters`, but for a distributed calibration `res[1]` is the flat vector the search ran over, not the `(rows, cols, n)` array a run reads -- two shapes describing different things under one name, which left the model unrunnable after a calibration. mypy caught this the moment `parameters` became a typed object. The run layer's read surface drops from 17 named attributes to 12, and the protocol conflicts against `Catchment` are now only `| None` on assembled objects -- the builder-versus-finished-model question -- with nothing smeared left. BREAKING CHANGE: `Catchment.parameters` is a `ParameterSet`, not an array -- read `model.parameters.values`. `snow` and `maxbas` are on it. `lumped_model`, `area`, `initial_cond` and `q_init` move to `Catchment.model_setup`. `LB`/`UB` become `Catchment.bounds.lower`/`.upper`. `Calibration.parameters` no longer holds the optimiser result; use `best_parameters`. --- pyproject.toml | 3 +- src/hapi/calibration.py | 61 +++- src/hapi/catchment.py | 87 ++--- src/hapi/conceptual.py | 304 ++++++++++++++++++ src/hapi/period.py | 6 +- src/hapi/protocols.py | 31 +- src/hapi/rrm/distrrm.py | 20 +- src/hapi/run.py | 2 +- src/hapi/wrapper.py | 26 +- tests/calibration/distributed_mode_calib.py | 4 +- tests/calibration/lumped_calibration.py | 2 +- .../test_calibration_distributed.py | 50 +-- tests/rrm/calibration/test_rrm_calibration.py | 8 +- tests/rrm/catchment/test_config.py | 18 +- tests/rrm/catchment/test_fw1_output_fields.py | 2 +- .../catchment/test_maxbas_routing_variants.py | 2 +- .../test_read_parameters_validation.py | 24 +- tests/rrm/catchment/test_rrm_catchment.py | 30 +- tests/rrm/catchment/test_run_validation.py | 7 +- tests/rrm/catchment/test_wrapper_with_lake.py | 4 +- tests/sensitivity_analysis.py | 13 +- 21 files changed, 513 insertions(+), 191 deletions(-) create mode 100644 src/hapi/conceptual.py diff --git a/pyproject.toml b/pyproject.toml index 9adfab27..d06f4d53 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -261,7 +261,8 @@ description = "Run all test suite" cmd = [ "pytest", "--doctest-modules", "-p", "no:cacheprovider", "--no-cov", "src/hapi/config.py", "src/hapi/catchment.py", - "src/hapi/inputs.py", "src/hapi/period.py", "src/hapi/results.py", + "src/hapi/conceptual.py", "src/hapi/inputs.py", "src/hapi/period.py", + "src/hapi/results.py", "src/hapi/routing.py", "src/hapi/run.py", ] diff --git a/src/hapi/calibration.py b/src/hapi/calibration.py index cbb19578..c0d4bf45 100644 --- a/src/hapi/calibration.py +++ b/src/hapi/calibration.py @@ -16,6 +16,7 @@ from Oasis.optimization import Optimization from hapi.catchment import Catchment +from hapi.conceptual import ParameterSet from hapi.wrapper import Wrapper ROWS_MISMATCH_ERROR = "all input data should have the same number of rows" @@ -107,6 +108,13 @@ def __init__( self.objective_function: Callable[..., Any] | None = None self.OFArgs: list | None = None self.OFvalue: float | None = None + #: The optimiser's answer -- the flat vector it searched over, as returned in + #: `res[1]`. Deliberately *not* the model's runnable parameter set: for a distributed + #: calibration the winning vector still has to go through the spatial-distribution + #: function to become the `(rows, cols, n)` array a run reads, so the two are + #: different shapes describing different things. The runnable set lives on + #: `self.parameters.values`. + self.best_parameters: np.ndarray | list | None = None def _declare_the_parameter_variables( self, opt_prob: Optimization, initial_values: list | None = None @@ -128,18 +136,45 @@ def _declare_the_parameter_variables( # part-way through building the problem, naming neither argument and leaving # `opt_prob` half-populated. seeded = initial_values is not None and len(initial_values) > 0 - if seeded and len(initial_values) != len(self.LB): + if seeded and len(initial_values) != len(self.bounds): raise ValueError( f"initial_values must hold one value per parameter; the bounds define " - f"{len(self.LB)} and {len(initial_values)} were given" + f"{len(self.bounds)} and {len(initial_values)} were given" ) - for i in range(len(self.LB)): + for i in range(len(self.bounds)): seed = {"value": initial_values[i]} if seeded else {} opt_prob.addVar( - f"x{i}", type="c", lower=self.LB[i], upper=self.UB[i], **seed + f"x{i}", + type="c", + lower=self.bounds.lower[i], + upper=self.bounds.upper[i], + **seed, ) + def _parameter_set(self, values) -> ParameterSet: + """Wrap a trial vector as a checked `ParameterSet`. + + The width rule needs the `(snow, maxbas)` pair, which a calibration supplies through + `read_parameters_bound` rather than by reading a parameter file. Either source works; + this picks whichever ran. + + Args: + values: The trial parameter array or vector. + + Returns: + ParameterSet: The set, its width checked against the configuration. + + Raises: + ValueError: The trial set is not the width the configuration requires. + """ + if self.parameters is not None: + return self.parameters.with_values(values) + bounds = self.bounds + snow = bounds.snow if bounds is not None else False + maxbas = bounds.maxbas if bounds is not None else False + return ParameterSet(values, snow=snow, maxbas=maxbas) + def read_objective_function( self, objective_function: Callable[..., Any], args: list | None ): @@ -317,7 +352,10 @@ def opt_fun(par): spatial_var_fun.Function( par ) # , kub=spatial_var_fun.Kub, klb=spatial_var_fun.Klb - self.parameters = spatial_var_fun.Par3d + # Re-checked per trial: `with_parameters` re-runs the (snow, maxbas) count + # rule, so a distribution function producing the wrong width fails here + # rather than as an index error inside the per-cell loop. + self.parameters = self._parameter_set(spatial_var_fun.Par3d) # run the model Wrapper.run_muskingum(self) # calculate performance of the model @@ -377,7 +415,7 @@ def opt_fun(par): hot_start=hot_start, ) - self.parameters = res[1] + self.best_parameters = res[1] self.OFvalue = res[0] return res @@ -463,7 +501,8 @@ def opt_fun(par): spatial_var_fun.Function( par ) # , kub=spatial_var_fun.Kub, klb=spatial_var_fun.Klb, Maskingum=spatial_var_fun.Maskingum - self.parameters = spatial_var_fun.Par3d + # Re-checked per trial -- see run_calibration. + self.parameters = self._parameter_set(spatial_var_fun.Par3d) # run the model Wrapper.run_maxbas(self) # calculate performance of the model @@ -508,7 +547,7 @@ def opt_fun(par): hot_start=hot_start, ) - self.parameters = res[1] + self.best_parameters = res[1] self.OFvalue = res[0] return res @@ -596,8 +635,8 @@ def calibrate_lumped( ### calculate the objective function def opt_fun(par): try: - # parameters - self.parameters = par + # parameters. Checked against (snow, maxbas) as it arrives. + self.parameters = self._parameter_set(par) # run the model Wrapper.run_lumped(self, route, routing_fn) # calculate performance of the model @@ -658,6 +697,6 @@ def opt_fun(par): ) self.OFvalue = res[0] - self.parameters = res[1] + self.best_parameters = res[1] return res diff --git a/src/hapi/catchment.py b/src/hapi/catchment.py index 714e8e89..fd4df040 100644 --- a/src/hapi/catchment.py +++ b/src/hapi/catchment.py @@ -35,6 +35,7 @@ from pyramids.dataset import DatasetCollection as Datacube from pyramids.feature import FeatureCollection +from hapi.conceptual import ConceptualModelSetup, ParameterBounds, ParameterSet from hapi.config import RunConfig from hapi.inputs import ( METEO_VARIABLES, @@ -300,21 +301,22 @@ def __init__( f"got {routing_method!r}" ) self.routing_method = ROUTING_METHODS[routing_method.lower()] - self.parameters: np.ndarray | list | None = None + #: The parameters and the `(snow, maxbas)` pair that fixes their width, as + #: `read_parameters` produces them. Its constructor enforces the count rule, so every + #: route to a parameter set is checked -- including the per-trial replacements a + #: calibration makes. `None` until read. + self.parameters: ParameterSet | None = None + #: The conceptual model and the state it starts from, as `read_lumped_model` + #: produces them. `None` until read. + self.model_setup: ConceptualModelSetup | None = None self.data: np.ndarray | None = None #: The three meteorological drivers. Assign a :class:`~hapi.inputs.MeteoInputs` #: built by one of its loaders; everything meteorological hangs off it. self.meteo: MeteoInputs | None = None self.QGauges: pd.DataFrame | None = None - self.snow: int | None = None - self.maxbas: bool | None = None - self.lumped_model: BaseConceptualModel | None = None - self.area: float | int | None = None - self.initial_cond: list | None = None - self.q_init: float | None = None self.GaugesTable: FeatureCollection | pd.DataFrame | None = None - self.UB: np.ndarray | None = None - self.LB: np.ndarray | None = None + #: The search space a calibration explores, once read. `None` otherwise. + self.bounds: ParameterBounds | None = None #: The routing network and the grid it defines. Assign a #: :class:`~hapi.inputs.FlowNetwork` built by its loader. self.flow_network: FlowNetwork | None = None @@ -645,37 +647,23 @@ def read_parameters(self, path: str, snow: bool = False, maxbas: bool = False): # offending directory, so _name_the_path re-raises with it. with _name_the_path(path): cube = read_rasters(path, regex_string=r"\d+", date=False) - self.parameters = np.moveaxis(cube.values, 0, -1) + parameters = np.moveaxis(cube.values, 0, -1) else: if not os.path.exists(path): raise FileNotFoundError( "The parameter file you have entered does not exist" ) - self.parameters = pd.read_csv(path, index_col=0, header=None)[1].tolist() + parameters = pd.read_csv(path, index_col=0, header=None)[1].tolist() if not (not snow or snow): raise ValueError( "snow input defines whether to consider snow subroutine or not it has to be True or False" ) - self.snow = snow - self.maxbas = maxbas - - # (snow, maxbas) -> the parameter count that combination requires. A table rather - # than eight near-identical branches: the counts are the only thing that varied, and - # two of the branches spelled the comparison `not len(...) == N`. - expected = PARAMETER_COUNTS[(bool(snow), bool(maxbas))] - actual = ( - self.parameters.shape[2] - if self.spatial_resolution == "distributed" - else len(self.parameters) - ) - if actual != expected: - raise ValueError( - f"current version of HBV (with snow) takes {expected} parameters you have " - f"entered {actual}" - ) + # The count check lives in `ParameterSet.__post_init__`, so it runs on every route + # to a parameter set rather than only on this one. + self.parameters = ParameterSet(parameters, snow=snow, maxbas=maxbas) logger.debug("Parameters are read successfully") @@ -712,29 +700,11 @@ def read_lumped_model( "ConceptualModel should be a module or a python file contains functions " ) - self.lumped_model = lumped_model() - self.area = catchment_area - - # Typed before it is measured. The other order called `len` first, so None reported - # "object of type 'NoneType' has no len()" rather than naming the argument, and the - # `is not None` the type check then carried could never be false -- `len` would - # already have raised. - if not isinstance(initial_condition, list): - raise TypeError( - f"init_st should be of type list, got {type(initial_condition).__name__}" - ) - if len(initial_condition) != 5: - raise ValueError( - f"state variables are 5 and the given initial values are {len(initial_condition)}" - ) - - self.initial_cond = initial_condition - - if q_init is not None and not isinstance(q_init, float): - raise TypeError( - f"q_init should be of type float, got {type(q_init).__name__}" - ) - self.q_init = q_init + # The checks on `initial_condition` and `q_init` live in + # `ConceptualModelSetup.__post_init__` now. + self.model_setup = ConceptualModelSetup( + lumped_model(), catchment_area, initial_condition, q_init + ) logger.debug("Lumped model is read successfully") @@ -1090,20 +1060,15 @@ def read_parameters_bound( `lower_bound` are not equal. ValueError: If `snow` is not a boolean. """ - if len(upper_bound) != len(lower_bound): - raise ValueError( - f"the length of UB should be the same as LB, got {len(upper_bound)} and " - f"{len(lower_bound)}" - ) - self.UB = np.array(upper_bound) - self.LB = np.array(lower_bound) - if not isinstance(snow, bool): raise ValueError( " snow input defines whether to consider snow subroutine or not it has to be True or False" ) - self.snow = snow - self.maxbas = maxbas + # A calibration reads no parameter file, so the bounds are where `(snow, maxbas)` + # enters -- carried here so every trial vector can be checked against it. + self.bounds = ParameterBounds( + lower_bound, upper_bound, snow=snow, maxbas=maxbas + ) logger.debug("Parameters' bounds are read successfully") diff --git a/src/hapi/conceptual.py b/src/hapi/conceptual.py new file mode 100644 index 00000000..75a352d8 --- /dev/null +++ b/src/hapi/conceptual.py @@ -0,0 +1,304 @@ +"""The conceptual model a run executes, and the parameters it is configured with. + +Seven attributes on :class:`~hapi.catchment.Catchment` described one thing -- the +rainfall-runoff model, configured and ready to run -- and were set by two readers that did +not know about each other: `read_parameters` set `parameters`, `snow` and `maxbas`, while +`read_lumped_model` set `lumped_model`, `area`, `initial_cond` and `q_init`. + +That mattered because `snow` and `maxbas` are not tags. They *determine how many parameters +there must be* (:data:`PARAMETER_COUNTS`), and that rule was enforced in exactly one place: +inside `read_parameters`, on the one assignment a calibration never makes. A calibration +replaces the parameter array once per trial vector, and none of those replacements were +checked -- so a spatial-distribution function producing the wrong width reached the per-cell +loop, where it fails as an index error far from the call that caused it. + +:class:`ParameterSet` holds the array with the `(snow, maxbas)` pair that fixes its width and +enforces the rule in its constructor, so every route to a parameter set goes through the same +check. :class:`ConceptualModelSetup` holds the other half. They are two objects rather than +one because they are read independently, in either order -- pairing them would force a +half-built object to exist, which is the thing being removed. +""" + +from __future__ import annotations + +from dataclasses import dataclass, replace +from typing import TYPE_CHECKING, Any + +import numpy as np + +if TYPE_CHECKING: + from hapi.rrm.base_model import BaseConceptualModel + +#: (snow, maxbas) -> how many parameters the conceptual model reads in that configuration. +#: The snow routine adds five; MAXBAS replaces the two Muskingum parameters with one. +PARAMETER_COUNTS: dict[tuple[bool, bool], int] = { + (True, True): 16, + (False, True): 11, + (True, False): 17, + (False, False): 12, +} + + +def parameter_count(parameters: np.ndarray | list) -> int: + """Count the parameters a set carries, whatever shape it is stored in. + + A distributed set is a `(rows, cols, n)` array and a lumped one a flat sequence, so the + count is read from the shape rather than from a mode flag the caller has to supply -- + which is how the check used to depend on `spatial_resolution`. + + Args: + parameters: The parameter set, 3D for distributed or 1D for lumped. + + Returns: + int: Number of parameters per cell (distributed) or in total (lumped). + + Examples: + ```python + >>> import numpy as np + >>> from hapi.conceptual import parameter_count + >>> parameter_count(np.zeros((13, 14, 12))) + 12 + >>> parameter_count([1.0] * 12) + 12 + + ``` + """ + array = np.asarray(parameters) + return array.shape[2] if array.ndim == 3 else len(array) + + +def validate_parameter_count( + parameters: np.ndarray | list, snow: bool, maxbas: bool +) -> None: + """Check a parameter set carries the count `(snow, maxbas)` requires. + + Split out of the spec's constructor so a builder can run it the moment the parameters and + the configuration are both known -- which is inside `read_parameters`, before the + conceptual model itself has been read. Waiting for the whole spec would move a + wrong-width error to whichever later call completed it. + + Args: + parameters: The parameter set. + snow: Whether the snow routine runs. + maxbas: Whether the set carries a MAXBAS value. + + Raises: + ValueError: The count does not match. + + Examples: + ```python + >>> import numpy as np + >>> from hapi.conceptual import validate_parameter_count + >>> validate_parameter_count(np.zeros((2, 2, 5)), snow=False, maxbas=False) + Traceback (most recent call last): + ... + ValueError: a model with snow=False, maxbas=False takes 12 parameters, got 5 + + ``` + """ + expected = PARAMETER_COUNTS[(bool(snow), bool(maxbas))] + actual = parameter_count(parameters) + if actual != expected: + raise ValueError( + f"a model with snow={bool(snow)}, maxbas={bool(maxbas)} takes {expected} " + f"parameters, got {actual}" + ) + + +def validate_initial_cond(initial_cond: Any) -> list: + """Check the initial state is the five values the conceptual models read. + + Exposed as a function, not only as part of the spec's constructor, so a builder can check + this half as it arrives rather than waiting for the other half to turn up -- which is what + keeps the error at the `read_lumped_model` call that supplied it. + + Args: + initial_cond: The candidate initial state. + + Returns: + list: `initial_cond`, unchanged. + + Raises: + TypeError: It is not a list. + ValueError: It does not hold exactly five values. + """ + if not isinstance(initial_cond, list): + raise TypeError( + f"init_st should be of type list, got {type(initial_cond).__name__}" + ) + if len(initial_cond) != 5: + raise ValueError( + f"state variables are 5 and the given initial values are {len(initial_cond)}" + ) + return initial_cond + + +def validate_q_init(q_init: Any) -> float | None: + """Check the initial discharge is a float or absent. + + Args: + q_init: The candidate initial discharge. + + Returns: + float | None: `q_init`, unchanged. + + Raises: + TypeError: It is neither None nor a float. + """ + if q_init is not None and not isinstance(q_init, float): + raise TypeError(f"q_init should be of type float, got {type(q_init).__name__}") + return q_init + + +@dataclass(frozen=True) +class ParameterSet: + """A parameter set together with the configuration that fixes its width. + + Exactly what `read_parameters` produces. `snow` and `maxbas` are not tags travelling + beside the array -- they *determine how many parameters there must be* + (:data:`PARAMETER_COUNTS`), so holding the three together is what lets the rule be checked + on construction instead of at one call site. + + Frozen: a calibration explores parameter sets by the thousand, and each is a different + set rather than a mutation of the last. :meth:`with_values` returns a new one and re-runs + the check, so a distribution function producing the wrong width fails on the trial that + produced it rather than as an index error inside the per-cell loop. + + Attributes: + values: `(rows, cols, n)` for a distributed run, a flat sequence for a lumped one. + snow: Whether the snow routine runs. + maxbas: Whether the set carries a MAXBAS value instead of Muskingum's two. + + Examples: + - The width has to match the configuration: + ```python + >>> import numpy as np + >>> from hapi.conceptual import ParameterSet + >>> ParameterSet(np.zeros((2, 2, 12))).count + 12 + >>> ParameterSet(np.zeros((2, 2, 5))) + Traceback (most recent call last): + ... + ValueError: a model with snow=False, maxbas=False takes 12 parameters, got 5 + + ``` + """ + + values: np.ndarray | list + snow: bool = False + maxbas: bool = False + + def __post_init__(self): + """Check the width matches `(snow, maxbas)`. + + Raises: + ValueError: The count does not match. + """ + validate_parameter_count(self.values, self.snow, self.maxbas) + + @property + def count(self) -> int: + """int: Number of parameters the set carries. See :func:`parameter_count`.""" + return parameter_count(self.values) + + def with_values(self, values: np.ndarray | list) -> ParameterSet: + """Return the same configuration with a different parameter array. + + Args: + values: The replacement parameter array. + + Returns: + ParameterSet: A new set carrying `values`, width already checked. + + Raises: + ValueError: The replacement does not carry the required count. + """ + return replace(self, values=values) + + +@dataclass(frozen=True) +class ConceptualModelSetup: + """The conceptual model instance and the state it starts from. + + Exactly what `read_lumped_model` produces. Kept apart from :class:`ParameterSet` because + the two are read independently and either may be read first -- pairing them would force a + half-built object to exist, which is the thing this refactor is removing. + + Attributes: + model: The conceptual model instance whose `simulate` is called per cell. + area: Catchment area in km2. + initial_cond: Initial state values `[sp, sm, uz, lz, wc]`. + q_init: Initial discharge in m3/s, or None to let the model choose. + + Examples: + ```python + >>> from hapi.conceptual import ConceptualModelSetup + >>> setup = ConceptualModelSetup(None, 1530.0, [0, 10, 10, 10, 0], q_init=5.0) + >>> setup.area, setup.q_init + (1530.0, 5.0) + + ``` + """ + + model: BaseConceptualModel | None + area: float | int + initial_cond: list + q_init: float | None = None + + def __post_init__(self): + """Check the initial state and discharge. + + Raises: + TypeError: `initial_cond` is not a list, or `q_init` is neither None nor a float. + ValueError: `initial_cond` does not hold five values. + """ + validate_initial_cond(self.initial_cond) + validate_q_init(self.q_init) + +@dataclass(frozen=True) +class ParameterBounds: + """The search space a calibration explores, and the configuration it explores it under. + + Exactly what `read_parameters_bound` produces. It carries the same `(snow, maxbas)` pair + as :class:`ParameterSet` because a calibration has no parameter file to read it from -- + the bounds are where the configuration enters, and every trial vector the optimiser + produces is checked against it. + + Attributes: + lower: Lower bound per parameter. + upper: Upper bound per parameter. + snow: Whether the snow routine runs. + maxbas: Whether the parameter vector carries a MAXBAS value. + + Examples: + ```python + >>> from hapi.conceptual import ParameterBounds + >>> bounds = ParameterBounds([0.0] * 12, [1.0] * 12) + >>> len(bounds) + 12 + + ``` + """ + + lower: np.ndarray | list + upper: np.ndarray | list + snow: bool = False + maxbas: bool = False + + def __post_init__(self): + """Check the two bounds describe the same parameters. + + Raises: + ValueError: The bounds are different lengths. + """ + if len(self.lower) != len(self.upper): + raise ValueError( + f"the length of UB should be the same as LB, got {len(self.upper)} and " + f"{len(self.lower)}" + ) + object.__setattr__(self, "lower", np.array(self.lower)) + object.__setattr__(self, "upper", np.array(self.upper)) + + def __len__(self) -> int: + """int: Number of parameters being calibrated.""" + return len(self.lower) diff --git a/src/hapi/period.py b/src/hapi/period.py index 53c00a2f..b3db1068 100644 --- a/src/hapi/period.py +++ b/src/hapi/period.py @@ -165,8 +165,10 @@ def days(self) -> int: @property def conversion_factor(self) -> float: """float: Depth-to-discharge factor -- mm over the catchment to m3/s at this step.""" - return CONVERSION_FACTOR if self.temporal_resolution == "daily" else ( - CONVERSION_FACTOR / 24 + return ( + CONVERSION_FACTOR + if self.temporal_resolution == "daily" + else (CONVERSION_FACTOR / 24) ) @property diff --git a/src/hapi/protocols.py b/src/hapi/protocols.py index 2dcd4ea3..17e841c2 100644 --- a/src/hapi/protocols.py +++ b/src/hapi/protocols.py @@ -17,17 +17,15 @@ from __future__ import annotations -from typing import TYPE_CHECKING, Any, Protocol +from typing import Any, Protocol import numpy as np +from hapi.conceptual import ConceptualModelSetup, ParameterSet from hapi.inputs import FlowNetwork, MeteoInputs from hapi.period import SimulationPeriod from hapi.results import SimulationResults -if TYPE_CHECKING: - from hapi.rrm.base_model import BaseConceptualModel - class ConceptualModelInputs(Protocol): """What the per-cell conceptual model needs, whatever routes its output. @@ -36,13 +34,10 @@ class ConceptualModelInputs(Protocol): :class:`LumpedModel` does not advertise a need for a flow network it never touches. Attributes: - parameters: The conceptual model's parameters. A 3D `(rows, cols, n)` array for a - distributed run; a flat sequence for a lumped one. - lumped_model: The conceptual model instance whose `simulate` is called per cell. - initial_cond: Initial state values `[sp, sm, uz, lz, wc]`. - q_init: Initial discharge in m3/s, or None to let the model choose. - snow: 1 to run the snow routine, 0 otherwise. - area: Catchment area in km2. + parameters: The parameter array together with the `(snow, maxbas)` pair that fixes + its width -- one object, so the width rule is checked on every route to a set. + model_setup: The conceptual model instance, the catchment area and the state the + run starts from. period: The span the run covers, and the calendar, `dt` and `conversion_factor` it implies. One object rather than six loose fields, so a derived value can never describe a different span from the one the model is set to. @@ -50,12 +45,8 @@ class ConceptualModelInputs(Protocol): """ period: SimulationPeriod - parameters: np.ndarray | list - lumped_model: BaseConceptualModel - initial_cond: list - q_init: float | None - snow: int - area: float | int + parameters: ParameterSet + model_setup: ConceptualModelSetup results: SimulationResults | None @@ -81,13 +72,11 @@ class LumpedModelInputs(ConceptualModelInputs, Protocol): Attributes: data: `(time, 4)` array of precipitation, ET, temperature and the long-term average. - maxbas: Whether the parameter vector carries a MAXBAS value, which changes how the - routing function is called. - Qsim: Where the routed hydrograph lands. + Qsim: Where the routed hydrograph lands. Whether MAXBAS routing applies is read off + `parameters.maxbas`, since that is what fixes the vector's width too. """ data: np.ndarray - maxbas: bool Qsim: Any diff --git a/src/hapi/rrm/distrrm.py b/src/hapi/rrm/distrrm.py index 4f760dc5..3fb916c1 100644 --- a/src/hapi/rrm/distrrm.py +++ b/src/hapi/rrm/distrrm.py @@ -100,18 +100,18 @@ def run_lumped_model(Model) -> SimulationResults: results.quz[x, y, :], results.qlz[x, y, :], results.state_variables[x, y, :, :], - ) = Model.lumped_model.simulate( + ) = Model.model_setup.model.simulate( prec=Model.meteo.precipitation[x, y, :], temp=Model.meteo.temperature[x, y, :], et=Model.meteo.evapotranspiration[x, y, :], ll_temp=Model.meteo.ll_temp[x, y, :], - par=Model.parameters[x, y, :], - init_st=Model.initial_cond, - q_init=Model.q_init, - snow=Model.snow, + par=Model.parameters.values[x, y, :], + init_st=Model.model_setup.initial_cond, + q_init=Model.model_setup.q_init, + snow=Model.parameters.snow, ) - area_coef = Model.area / Model.flow_network.px_tot_area + area_coef = Model.model_setup.area / Model.flow_network.px_tot_area factor = Model.flow_network.px_area * area_coef / Model.period.conversion_factor # convert quz and qlz from mm/time step to m3/sec # Timef*3.6 results.quz = results.quz * factor @@ -222,8 +222,8 @@ def route_muskingum(Model): q_uzi = q_uzi + routing.muskingum_v( results.quz_routed[x_ind, y_ind, :], results.quz_routed[x_ind, y_ind, 0], - Model.parameters[x_ind, y_ind, 10], - Model.parameters[x_ind, y_ind, 11], + Model.parameters.values[x_ind, y_ind, 10], + Model.parameters.values[x_ind, y_ind, 11], Model.period.dt, ) @@ -263,7 +263,7 @@ def route_maxbas(Model): - `quz` (numpy.ndarray): 3-D upper-zone discharge array `(rows, cols, TS)` in m3/s. """ - Maxbas = Model.parameters[:, :, -1] + Maxbas = Model.parameters.values[:, :, -1] quz = Model.results.quz for x in range(Model.flow_network.rows): @@ -300,7 +300,7 @@ def route_maxbas_by_path_length(Model): - `quz` (numpy.ndarray): 3-D upper-zone discharge array `(rows, cols, TS)` in m3/s. """ - MAXBAS = np.nanmax(Model.parameters[:, :, -1]) + MAXBAS = np.nanmax(Model.parameters.values[:, :, -1]) # `read_flow_path_length` already masks this raster's own no-data cells to NaN via # pyramids, so no sentinel comparison is needed here -- and the one that used to sit # here compared against the *accumulation* raster's sentinel, which is a different diff --git a/src/hapi/run.py b/src/hapi/run.py index f1c9bfbc..610543b7 100644 --- a/src/hapi/run.py +++ b/src/hapi/run.py @@ -54,7 +54,7 @@ def _check_parameters_cover_grid(model: DistributedModel) -> None: Raises: ValueError: The parameter array has the wrong number of rows or columns. """ - shape = np.asarray(model.parameters).shape + shape = np.asarray(model.parameters.values).shape if shape[0] != model.flow_network.rows: raise ValueError(ROWS_MISMATCH_ERROR) if shape[1] != model.flow_network.cols: diff --git a/src/hapi/wrapper.py b/src/hapi/wrapper.py index 719d85fa..e72ac006 100644 --- a/src/hapi/wrapper.py +++ b/src/hapi/wrapper.py @@ -160,8 +160,8 @@ def run_muskingum_with_lake( qlake = routing.muskingum_v( Lake.QlakeR, Lake.QlakeR[0], - Model.parameters[Lake.OutflowCell[0], Lake.OutflowCell[1], 10], - Model.parameters[Lake.OutflowCell[0], Lake.OutflowCell[1], 11], + Model.parameters.values[Lake.OutflowCell[0], Lake.OutflowCell[1], 10], + Model.parameters.values[Lake.OutflowCell[0], Lake.OutflowCell[1], 11], Model.period.conversion_factor, ) @@ -416,19 +416,19 @@ def run_lumped( tm = Model.data[:, 3] # from the conceptual model calculate the upper and lower response mm/time step - quz, qlz, state_variables = Model.lumped_model.simulate( + quz, qlz, state_variables = Model.model_setup.model.simulate( p, t, et, tm, - Model.parameters, - init_st=Model.initial_cond, - q_init=Model.q_init, - snow=Model.snow, + Model.parameters.values, + init_st=Model.model_setup.initial_cond, + q_init=Model.model_setup.q_init, + snow=Model.parameters.snow, ) # q mm , area sq km (1000**2)/1000/f/60/60 = 1/(3.6*f) # if daily tfac=24 if hourly tfac=1 if 15 min tfac=0.25 - factor = Model.area / Model.period.conversion_factor + factor = Model.model_setup.area / Model.period.conversion_factor # A lumped run has no spatial routing at all, so the routed fields stay None and # the routing kind says why -- rather than a MAXBAS flag left over from elsewhere. results = SimulationResults( @@ -441,14 +441,16 @@ def run_lumped( Model.Qsim = results.quz + results.qlz - if Routing != 0 and Model.maxbas: - Model.Qsim = RoutingFn(np.array(Model.Qsim[:-1]), Model.parameters[-1]) + if Routing != 0 and Model.parameters.maxbas: + Model.Qsim = RoutingFn( + np.array(Model.Qsim[:-1]), Model.parameters.values[-1] + ) elif Routing != 0: Model.Qsim = RoutingFn( np.array(Model.Qsim[:-1]), Model.Qsim[0], - Model.parameters[-2], - Model.parameters[-1], + Model.parameters.values[-2], + Model.parameters.values[-1], Model.period.dt, ) return results diff --git a/tests/calibration/distributed_mode_calib.py b/tests/calibration/distributed_mode_calib.py index 8576d677..77e33431 100644 --- a/tests/calibration/distributed_mode_calib.py +++ b/tests/calibration/distributed_mode_calib.py @@ -138,8 +138,8 @@ def objective_function(Qobs, Qout, q_uz_routed, q_lz_trans, coordinates): spatial_var_fun, optimization_args, print_error=0 ) # %% convert parameters to rasters -# Coello.parameters = [0.700, 399, 1.704, 0.1021, 0.4622, 0.6237, 0.1251, 0.005, 59.85, 5.241, 94.91, 0.2075] +# Coello.parameters.values = [0.700, 399, 1.704, 0.1021, 0.4622, 0.6237, 0.1251, 0.005, 59.85, 5.241, 94.91, 0.2075] spatial_var_fun.Function( - Coello.parameters, kub=spatial_var_fun.Kub, klb=spatial_var_fun.Klb + Coello.parameters.values, kub=spatial_var_fun.Kub, klb=spatial_var_fun.Klb ) spatial_var_fun.save_parameters(SaveTo) diff --git a/tests/calibration/lumped_calibration.py b/tests/calibration/lumped_calibration.py index d5176e30..4d9242eb 100644 --- a/tests/calibration/lumped_calibration.py +++ b/tests/calibration/lumped_calibration.py @@ -95,7 +95,7 @@ print("Parameters are " + str(cal_parameters[1])) print("Time = " + str(round(cal_parameters[2]["time"] / 60, 2)) + " min") # %% run the model -Coello.parameters = cal_parameters[1] +Coello.parameters.values = cal_parameters[1] Run.run_lumped(Coello, Route, routing_fn) # %% calculate performance criteria scores = dict() diff --git a/tests/rrm/calibration/test_calibration_distributed.py b/tests/rrm/calibration/test_calibration_distributed.py index 0e53a884..57be2bc8 100644 --- a/tests/rrm/calibration/test_calibration_distributed.py +++ b/tests/rrm/calibration/test_calibration_distributed.py @@ -14,6 +14,7 @@ from hapi import calibration as calibration_module from hapi.calibration import Calibration +from hapi.conceptual import ParameterBounds from hapi.inputs import FlowNetwork, MeteoInputs from hapi.results import RoutingKind, SimulationResults from hapi.routing import Routing @@ -206,8 +207,7 @@ def test_stores_the_optimizer_result_on_the_instance( """ coello = gauged_calibration coello.read_objective_function(metrics.rmse, []) - coello.LB = np.zeros(12) - coello.UB = np.ones(12) + coello.bounds = ParameterBounds(np.zeros(12), np.ones(12)) res = coello.run_calibration(spatial_var_stub, _optimization_args()) @@ -216,9 +216,12 @@ def test_stores_the_optimizer_result_on_the_instance( f"OFvalue must be res[0], got {coello.OFvalue}" ) np.testing.assert_array_equal( - coello.parameters, + coello.best_parameters, CANNED_RESULT[1], - err_msg="parameters must be res[1], lowercase — not a second attribute", + err_msg=( + "the optimiser's answer belongs on best_parameters: `parameters` is the " + "runnable ParameterSet, a different shape describing a different thing" + ), ) def test_rejects_meteo_that_does_not_cover_the_grid( @@ -232,8 +235,7 @@ def test_rejects_meteo_that_does_not_cover_the_grid( grids would burn a full optimisation before failing. """ coello = gauged_calibration - coello.LB = np.zeros(12) - coello.UB = np.ones(12) + coello.bounds = ParameterBounds(np.zeros(12), np.ones(12)) # Built in one go: replacing the cubes one at a time is now refused, because a # half-applied crop is exactly the inconsistency MeteoInputs guarantees against. coello.meteo = MeteoInputs( @@ -274,14 +276,13 @@ def test_the_objective_runs_the_model_on_the_trial_parameters( """ coello = gauged_calibration coello.read_objective_function(_pairwise_objective, []) - coello.LB = np.zeros(12) - coello.UB = np.ones(12) + coello.bounds = ParameterBounds(np.zeros(12), np.ones(12)) ran_with: list[np.ndarray] = [] original = calibration_module.Wrapper.run_muskingum def spy(model, *args, **kwargs): - ran_with.append(np.asarray(model.parameters, dtype=float).copy()) + ran_with.append(np.asarray(model.parameters.values, dtype=float).copy()) return original(model, *args, **kwargs) monkeypatch.setattr( @@ -329,8 +330,7 @@ def test_stores_the_optimizer_result_on_the_instance( """ coello = gauged_calibration coello.read_objective_function(metrics.rmse, []) - coello.LB = np.zeros(12) - coello.UB = np.ones(12) + coello.bounds = ParameterBounds(np.zeros(12), np.ones(12)) res = coello.calibrate_maxbas(spatial_var_stub, _optimization_args()) @@ -339,9 +339,12 @@ def test_stores_the_optimizer_result_on_the_instance( f"OFvalue must be res[0], got {coello.OFvalue}" ) np.testing.assert_array_equal( - coello.parameters, + coello.best_parameters, CANNED_RESULT[1], - err_msg="parameters must be res[1], lowercase — not a second attribute", + err_msg=( + "the optimiser's answer belongs on best_parameters: `parameters` is the " + "runnable ParameterSet, a different shape describing a different thing" + ), ) @@ -382,8 +385,7 @@ def test_a_non_dict_bundle_is_refused_before_the_optimizer_is_built( """ coello = gauged_calibration coello.read_objective_function(metrics.rmse, []) - coello.LB = np.zeros(12) - coello.UB = np.ones(12) + coello.bounds = ParameterBounds(np.zeros(12), np.ones(12)) with pytest.raises(TypeError, match=f"{bad_kind} arguments should be a dict"): coello.run_calibration(spatial_var_stub, args) @@ -411,8 +413,7 @@ def test_stores_the_optimizer_result_on_the_instance( """ coello = Calibration("rrm", coello_rrm_date[0], coello_rrm_date[1]) coello.read_lumped_inputs(lumped_meteo_data_path) - coello.LB = np.zeros(12) - coello.UB = np.ones(12) + coello.bounds = ParameterBounds(np.zeros(12), np.ones(12)) basic_inputs = dict( Route=0, RoutingFn=Routing.triangular_routing_1, InitialValues=[] ) @@ -424,9 +425,12 @@ def test_stores_the_optimizer_result_on_the_instance( f"OFvalue must be res[0], got {coello.OFvalue}" ) np.testing.assert_array_equal( - coello.parameters, + coello.best_parameters, CANNED_RESULT[1], - err_msg="parameters must be res[1], lowercase — not a second attribute", + err_msg=( + "the optimiser's answer belongs on best_parameters: `parameters` is the " + "runnable ParameterSet, a different shape describing a different thing" + ), ) def test_initial_values_are_seeded_into_the_problem( @@ -444,8 +448,7 @@ def test_initial_values_are_seeded_into_the_problem( """ coello = Calibration("rrm", coello_rrm_date[0], coello_rrm_date[1]) coello.read_lumped_inputs(lumped_meteo_data_path) - coello.LB = np.zeros(12) - coello.UB = np.ones(12) + coello.bounds = ParameterBounds(np.zeros(12), np.ones(12)) basic_inputs = dict( Route=0, RoutingFn=Routing.triangular_routing_1, @@ -472,15 +475,14 @@ def test_a_mismatched_initial_values_length_is_refused( stub_optimizer: Records whether the optimiser was reached. Test scenario: - The seeded branch loops `range(len(self.LB))` and indexes `initial_values[i]`, so + The seeded branch loops `range(len(self.bounds.lower))` and indexes `initial_values[i]`, so a shorter list used to run past its end partway through building the problem, leaving `opt_prob` half-populated and raising `IndexError` far from the call that supplied the list. The length is now compared up front. """ coello = Calibration("rrm", coello_rrm_date[0], coello_rrm_date[1]) coello.read_lumped_inputs(lumped_meteo_data_path) - coello.LB = np.zeros(12) - coello.UB = np.ones(12) + coello.bounds = ParameterBounds(np.zeros(12), np.ones(12)) basic_inputs = dict( Route=0, RoutingFn=Routing.triangular_routing_1, diff --git a/tests/rrm/calibration/test_rrm_calibration.py b/tests/rrm/calibration/test_rrm_calibration.py index 8ac25715..f53dd3f0 100644 --- a/tests/rrm/calibration/test_rrm_calibration.py +++ b/tests/rrm/calibration/test_rrm_calibration.py @@ -17,10 +17,10 @@ def test_read_parameters_bounds( Maxbas = True Snow = False Coello.read_parameters_bound(lower_bound, upper_bound, Snow, maxbas=Maxbas) - assert isinstance(Coello.UB, np.ndarray) - assert isinstance(Coello.LB, np.ndarray) - assert isinstance(Coello.snow, bool) - assert isinstance(Coello.maxbas, bool) + assert isinstance(Coello.bounds.upper, np.ndarray) + assert isinstance(Coello.bounds.lower, np.ndarray) + assert isinstance(Coello.bounds.snow, bool) + assert isinstance(Coello.bounds.maxbas, bool) def test_lumped_calibration( diff --git a/tests/rrm/catchment/test_config.py b/tests/rrm/catchment/test_config.py index f0da4fae..e73a2459 100644 --- a/tests/rrm/catchment/test_config.py +++ b/tests/rrm/catchment/test_config.py @@ -1322,13 +1322,15 @@ def test_a_distributed_configuration_populates_every_input( assert model.spatial_resolution == "distributed" assert model.meteo is not None, "meteo was not assigned" assert model.flow_network is not None, "flow_network was not assigned" - assert model.parameters is not None, "parameters were not read" - assert model.lumped_model is not None, "the conceptual model was not read" + assert model.parameters.values is not None, "parameters were not read" + assert model.model_setup.model is not None, "the conceptual model was not read" assert model.GaugesTable is not None, "the gauge table was not read" assert model.QGauges is not None, "the discharge was not read" - assert model.area == coello_cat_area, f"area not set: {model.area}" - assert model.initial_cond == coello_initial_cond, ( - f"initial condition not set: {model.initial_cond}" + assert model.model_setup.area == coello_cat_area, ( + f"area not set: {model.model_setup.area}" + ) + assert model.model_setup.initial_cond == coello_initial_cond, ( + f"initial condition not set: {model.model_setup.initial_cond}" ) def test_the_drivers_cover_the_model_period(self, distributed_mapping, tmp_path): @@ -1393,7 +1395,7 @@ def test_omitting_parameters_leaves_them_unread( model = Catchment.from_yaml(write_yaml(distributed_mapping, tmp_path)) assert model.parameters is None, ( - f"parameters should be unread, got {type(model.parameters)}" + f"parameters should be unread, got {type(model.parameters.values)}" ) assert model.meteo is not None, "the rest of the build should still have run" @@ -1544,7 +1546,9 @@ def test_a_lumped_configuration_reads_the_averaged_driver_csv( assert model.data is not None, "the averaged drivers were not read" assert model.meteo is None, "a lumped run should build no driver grid" assert model.flow_network is None, "a lumped run should build no flow network" - assert model.parameters is not None, "the lumped parameter file was not read" + assert model.parameters.values is not None, ( + "the lumped parameter file was not read" + ) def test_a_lumped_run_reads_discharge_without_a_gauge_table( self, lumped_mapping, tmp_path diff --git a/tests/rrm/catchment/test_fw1_output_fields.py b/tests/rrm/catchment/test_fw1_output_fields.py index c4fdbb81..d8ecc156 100644 --- a/tests/rrm/catchment/test_fw1_output_fields.py +++ b/tests/rrm/catchment/test_fw1_output_fields.py @@ -141,7 +141,7 @@ def test_fw1_qtot_matches_an_independent_triangular_convolution( wrong one. """ expected_quz = coello_unrouted.results.quz.copy() - maxbas = coello_fw1.parameters[:, :, -1] + maxbas = coello_fw1.parameters.values[:, :, -1] acc = coello_fw1.flow_network.flow_acc_arr for x in range(coello_fw1.flow_network.rows): for y in range(coello_fw1.flow_network.cols): diff --git a/tests/rrm/catchment/test_maxbas_routing_variants.py b/tests/rrm/catchment/test_maxbas_routing_variants.py index fe521dfa..572e09f6 100644 --- a/tests/rrm/catchment/test_maxbas_routing_variants.py +++ b/tests/rrm/catchment/test_maxbas_routing_variants.py @@ -245,7 +245,7 @@ def test_maxbas_routing_convolves_qsim_with_the_last_parameter( Run.run_lumped(routed, Route=1, routing_fn=Routing.triangular_routing_1) - maxbas = routed.parameters[-1] + maxbas = routed.parameters.values[-1] expected = Routing.triangular_routing_1( np.array(np.asarray(unrouted.Qsim)[:-1]), maxbas ) diff --git a/tests/rrm/catchment/test_read_parameters_validation.py b/tests/rrm/catchment/test_read_parameters_validation.py index b9b95a57..640ef399 100644 --- a/tests/rrm/catchment/test_read_parameters_validation.py +++ b/tests/rrm/catchment/test_read_parameters_validation.py @@ -182,13 +182,15 @@ def test_accepts_the_matching_configuration( distributed.read_parameters(path, snow, maxbas=maxbas) - assert distributed.parameters.shape[2] == expected_bands, ( + assert distributed.parameters.values.shape[2] == expected_bands, ( f"Expected {expected_bands} parameter bands, " - f"got {distributed.parameters.shape[2]}" + f"got {distributed.parameters.values.shape[2]}" ) - assert distributed.snow is snow, f"snow flag not stored: {distributed.snow}" - assert distributed.maxbas is maxbas, ( - f"maxbas flag not stored: {distributed.maxbas}" + assert distributed.parameters.snow is snow, ( + f"snow flag not stored: {distributed.parameters.snow}" + ) + assert distributed.parameters.maxbas is maxbas, ( + f"maxbas flag not stored: {distributed.parameters.maxbas}" ) def test_missing_directory_names_the_path_it_could_not_read( @@ -285,8 +287,8 @@ def test_a_float_initial_discharge_is_accepted( model.read_lumped_model(HBVLumped, 1530.0, coello_initial_cond, q_init=5.0) - assert model.q_init == pytest.approx(5.0), ( - f"the initial discharge must be stored, got {model.q_init}" + assert model.model_setup.q_init == pytest.approx(5.0), ( + f"the initial discharge must be stored, got {model.model_setup.q_init}" ) def test_omitting_the_initial_discharge_leaves_it_unset( @@ -302,8 +304,8 @@ def test_omitting_the_initial_discharge_leaves_it_unset( model.read_lumped_model(HBVLumped, 1530.0, coello_initial_cond) - assert model.q_init is None, ( - f"expected no initial discharge, got {model.q_init}" + assert model.model_setup.q_init is None, ( + f"expected no initial discharge, got {model.model_setup.q_init}" ) @pytest.mark.parametrize("bad", [5, "5.0", [5.0]], ids=["int", "str", "list"]) @@ -349,8 +351,8 @@ def test_a_list_of_five_is_accepted( model.read_lumped_model(HBVLumped, 1530.0, coello_initial_cond) - assert model.initial_cond == coello_initial_cond, ( - f"the initial condition must be stored unchanged, got {model.initial_cond}" + assert model.model_setup.initial_cond == coello_initial_cond, ( + f"the initial condition must be stored unchanged, got {model.model_setup.initial_cond}" ) @pytest.mark.parametrize( diff --git a/tests/rrm/catchment/test_rrm_catchment.py b/tests/rrm/catchment/test_rrm_catchment.py index bd9223d2..8b202d60 100644 --- a/tests/rrm/catchment/test_rrm_catchment.py +++ b/tests/rrm/catchment/test_rrm_catchment.py @@ -38,9 +38,9 @@ def test_read_lumped_model( ): coello = Catchment("rrm", coello_rrm_date[0], coello_rrm_date[1]) coello.read_lumped_model(HBVLumped, coello_AreaCoeff, coello_InitialCond) - assert isinstance(coello.lumped_model, HBVLumped) - assert isinstance(coello.area, float) - assert isinstance(coello.initial_cond, list) + assert isinstance(coello.model_setup.model, HBVLumped) + assert isinstance(coello.model_setup.area, float) + assert isinstance(coello.model_setup.initial_cond, list) def test_read_lumped_read_parameters( self, @@ -50,8 +50,8 @@ def test_read_lumped_read_parameters( ): coello = Catchment("rrm", coello_rrm_date[0], coello_rrm_date[1]) coello.read_parameters(lumped_parameters_path, coello_Snow) - assert isinstance(coello.parameters, list) - assert coello.snow == coello_Snow + assert isinstance(coello.parameters.values, list) + assert coello.parameters.snow == coello_Snow def test_read_discharge_gauges( self, @@ -223,9 +223,9 @@ def test_read_lumped_model( fmt="%Y-%m-%d", ) coello.read_lumped_model(HBVLumped, coello_cat_area, coello_initial_cond) - assert isinstance(coello.lumped_model, HBVLumped) - assert coello.area == coello_cat_area - assert coello.initial_cond == coello_initial_cond + assert isinstance(coello.model_setup.model, HBVLumped) + assert coello.model_setup.area == coello_cat_area + assert coello.model_setup.initial_cond == coello_initial_cond def test_read_parameters_bound( self, @@ -245,10 +245,10 @@ def test_read_parameters_bound( fmt="%Y-%m-%d", ) coello.read_parameters_bound(UB, LB, Snow) - assert all(coello.LB == LB) - assert all(coello.UB == UB) - assert coello.snow == Snow - assert coello.maxbas == False + assert all(coello.bounds.lower == LB) + assert all(coello.bounds.upper == UB) + assert coello.bounds.snow == Snow + assert coello.bounds.maxbas is False def test_read_gauge_table( self, @@ -312,13 +312,13 @@ def test_read_parameters_maxbas( ) Snow = False coello.read_parameters(coello_dist_parameters_maxbas, Snow, maxbas=True) - assert coello.parameters.shape == ( + assert coello.parameters.values.shape == ( coello_rows, coello_cols, coello_no_parameters - 1, ) - assert coello.snow == Snow - assert coello.maxbas is True + assert coello.parameters.snow == Snow + assert coello.parameters.maxbas is True class TestFW1: diff --git a/tests/rrm/catchment/test_run_validation.py b/tests/rrm/catchment/test_run_validation.py index 2ce6de00..6f3f0d9e 100644 --- a/tests/rrm/catchment/test_run_validation.py +++ b/tests/rrm/catchment/test_run_validation.py @@ -282,7 +282,12 @@ def test_rejects_parameters_off_the_catchment_grid( The parameter rasters are read independently of the GIS inputs, so they can disagree with the grid. A parameter cube one row short must raise. """ - coello_loaded.parameters = coello_loaded.parameters[:-1, :, :] + # Trimming a row keeps the parameter *width*, so `ParameterSet` still accepts it -- + # its rule is the count per cell. Covering the grid is the flow network's business, + # which is what `_check_parameters_cover_grid` is for. + coello_loaded.parameters = coello_loaded.parameters.with_values( + coello_loaded.parameters.values[:-1, :, :] + ) lake = _LakeStub(coello_loaded.meteo.time_steps) with pytest.raises(ValueError, match="as many rows as the catchment grid"): diff --git a/tests/rrm/catchment/test_wrapper_with_lake.py b/tests/rrm/catchment/test_wrapper_with_lake.py index bfd98ca1..43c678d9 100644 --- a/tests/rrm/catchment/test_wrapper_with_lake.py +++ b/tests/rrm/catchment/test_wrapper_with_lake.py @@ -314,8 +314,8 @@ def test_the_lake_raises_discharge_at_the_outflow_cell( expected = Routing.muskingum_v( lake.QlakeR, lake.QlakeR[0], - model.parameters[row, col, 10], - model.parameters[row, col, 11], + model.parameters.values[row, col, 10], + model.parameters.values[row, col, 11], model.period.conversion_factor, ) np.testing.assert_allclose( diff --git a/tests/sensitivity_analysis.py b/tests/sensitivity_analysis.py index cc2ac3c6..f3700f6b 100644 --- a/tests/sensitivity_analysis.py +++ b/tests/sensitivity_analysis.py @@ -111,7 +111,7 @@ # For Type 1 def WrapperType1(Randpar, Route, routing_fn, Qobs): - Coello.parameters = Randpar + Coello.parameters.values = Randpar Run.run_lumped(Coello, Route, routing_fn) rmse = metrics.rmse(Qobs, Coello.Qsim["q"]) @@ -120,7 +120,7 @@ def WrapperType1(Randpar, Route, routing_fn, Qobs): # For Type 2 def WrapperType2(Randpar, Route, routing_fn, Qobs): - Coello.parameters = Randpar + Coello.parameters.values = Randpar Run.run_lumped(Coello, Route, routing_fn) rmse = metrics.rmse(Qobs, Coello.Qsim["q"]) @@ -135,7 +135,14 @@ def WrapperType2(Randpar, Route, routing_fn, Qobs): fn = WrapperType2 -Sen = SA(parameters, Coello.LB, Coello.UB, fn, n_values=5, return_values=Type) +Sen = SA( + parameters, + Coello.bounds.lower, + Coello.bounds.upper, + fn, + n_values=5, + return_values=Type, +) Sen.one_at_a_time(Route, routing_fn, Qobs) # %% From = "" From e6f851c4353055c1f3b00baf41f0b9e6d609ea2e Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Wed, 2 Sep 2026 00:02:58 +0200 Subject: [PATCH 08/54] fix(routing): read the kinematic-wave skip at the entry point, not in the loop `route_muskingum` decided whether to skip a cell with if Model.routing_method != "Muskingum" and Model.bankfull_depth[x, y] > 0: which conflated two things. The intent is the flood model's: river cells are routed by the kinematic-wave model, so the Muskingum pass leaves them alone. But the comparison ran inside the routing loop on *every* distributed run, so a catchment declaring `routing_method="Kinematic"` and calling `Run.run_distributed` dereferenced `bankfull_depth` -- None outside the flood model -- and died with `TypeError: 'NoneType' object is not subscriptable` partway through routing. Confirmed against the Coello data: Muskingum completes, Kinematic crashes. The skip is now an explicit argument threaded from the entry point: `route_muskingum(Model, skip_hydraulic_cells=False)`, set by `Run.run_flood`, which derives it from `routing_method == "Kinematic"` and takes an override. `run_distributed` never skips, so the declared method cannot break it. `"Kinematic"` stays a valid routing method. It names the kinematic-wave scheme the flood model applies (`SaintVenant.KinematicRaster`, and the routing roadmap in README), it is documented as such in `docs/api/catchment.md`, and `config.py` already explains why YAML deliberately does not expose it. Only *where it is read* changes -- the entry point rather than the inner loop. `routing_method` also stays. It is not decoration: `hapi.config` cross-checks it against `parameters.maxbas`, and that check is load-bearing, because a MAXBAS set holds 11 parameters and a Muskingum set 12 while `maxbas` decides which count is expected -- so a set contradicting the routing passes the count check and the run then reads the Muskingum X as the MAXBAS value, giving a quietly wrong hydrograph. It is dropped from the `DistributedModel` protocol, since the distributed path no longer reads it, and added to `FloodModel`, which does. --- src/hapi/catchment.py | 27 +++++--- src/hapi/conceptual.py | 1 + src/hapi/protocols.py | 11 ++-- src/hapi/rrm/distrrm.py | 23 ++++--- src/hapi/run.py | 21 ++++-- src/hapi/wrapper.py | 13 +++- tests/rrm/catchment/test_config.py | 10 +-- .../catchment/test_run_results_coupling.py | 59 +++++++++++++++++ tests/rrm/catchment/test_run_validation.py | 65 ++++++++++++++++++- 9 files changed, 194 insertions(+), 36 deletions(-) diff --git a/src/hapi/catchment.py b/src/hapi/catchment.py index fd4df040..4eb4d8b3 100644 --- a/src/hapi/catchment.py +++ b/src/hapi/catchment.py @@ -72,13 +72,21 @@ "HBV": HBV, } -#: Accepted routing methods, mapped to the one spelling the internals compare against. -#: `distrrm.route_muskingum` tests `routing_method != "Muskingum"` exactly, so the constructor -#: canonicalises rather than storing what it was handed. `"Kinematic"` belongs here because -#: that comparison is also how the flood model selects its own path: a non-Muskingum method -#: with a real `bankfull_depth` skips the cell, which `Run.run_flood` relies on. -#: `hapi.config.CatchmentConfig.routing_method` exposes the first two to YAML and says why the -#: third is not; a method added here needs a decision there too. +#: Accepted routing methods, canonicalised to one spelling. +#: +#: This records *which routing the parameter set was calibrated for*. It does not select the +#: routing -- the `Run.*` entry point does that -- but it is not decoration either: +#: `hapi.config` cross-checks it against `parameters.maxbas`, and that check is load-bearing. +#: A MAXBAS set holds 11 parameters and a Muskingum set 12, and `maxbas` is what decides which +#: count is expected, so a set contradicting the routing still passes the count check and the +#: run then reads the Muskingum X as the MAXBAS value -- a quietly wrong hydrograph. +#: +#: `"Kinematic"` is the kinematic-wave routing the flood model applies to river cells (see +#: `SaintVenant.KinematicRaster`, and the roadmap item in README). `Run.run_flood` reads it to +#: decide whether the Muskingum pass should leave those cells alone. It is *read by the entry +#: point*, not compared inside the routing loop -- that comparison used to run on every +#: distributed model, so a catchment declaring Kinematic and calling `run_distributed` +#: dereferenced a `bankfull_depth` of None and crashed partway through routing. ROUTING_METHODS = { "muskingum": "Muskingum", "maxbas": "MAXBAS", @@ -290,9 +298,8 @@ def __init__( start_data, end, fmt=fmt, temporal_resolution=temporal_resolution ) - # Canonicalised like the two resolutions above, and for a sharper reason: - # `distrrm.route_muskingum` tests `routing_method != "Muskingum"` case-sensitively, and - # the false branch reads `bankfull_depth`, which is None outside the flood model. Left + # Canonicalised so the config cross-check against `parameters.maxbas` compares one + # spelling. The routing loop no longer compares against it at all. Left # verbatim, a lower-case "muskingum" therefore routed every cell down the MAXBAS branch # and raised `TypeError: 'NoneType' object is not subscriptable`. if routing_method.lower() not in ROUTING_METHODS: diff --git a/src/hapi/conceptual.py b/src/hapi/conceptual.py index 75a352d8..2e187bd8 100644 --- a/src/hapi/conceptual.py +++ b/src/hapi/conceptual.py @@ -255,6 +255,7 @@ def __post_init__(self): validate_initial_cond(self.initial_cond) validate_q_init(self.q_init) + @dataclass(frozen=True) class ParameterBounds: """The search space a calibration explores, and the configuration it explores it under. diff --git a/src/hapi/protocols.py b/src/hapi/protocols.py index 17e841c2..4f12f253 100644 --- a/src/hapi/protocols.py +++ b/src/hapi/protocols.py @@ -56,15 +56,10 @@ class DistributedModel(ConceptualModelInputs, Protocol): Attributes: meteo: The three driver cubes and the calendar they cover. flow_network: The routing network and the grid it defines. - routing_method: Canonicalised routing method. `route_muskingum` compares this against - `"Muskingum"` exactly to decide whether a cell is routed or skipped. - bankfull_depth: Read only when `routing_method` is not `"Muskingum"`; None otherwise. """ meteo: MeteoInputs flow_network: FlowNetwork - routing_method: str - bankfull_depth: np.ndarray | None class LumpedModelInputs(ConceptualModelInputs, Protocol): @@ -84,11 +79,17 @@ class FloodModel(DistributedModel, Protocol): """A distributed model that also carries the river geometry the flood model reads. Attributes: + routing_method: `"Kinematic"` when the kinematic-wave model routes the river cells, + which is what `run_flood` reads to decide whether the Muskingum pass skips them. + bankfull_depth: `(rows, cols)` bankfull depth. A positive value marks a river cell, + which the flood model can leave for a 1D hydraulic model to route. river_width: `(rows, cols)` channel width. river_roughness: `(rows, cols)` channel roughness. flood_plain_roughness: `(rows, cols)` floodplain roughness. """ + routing_method: str + bankfull_depth: np.ndarray river_width: np.ndarray river_roughness: np.ndarray flood_plain_roughness: np.ndarray diff --git a/src/hapi/rrm/distrrm.py b/src/hapi/rrm/distrrm.py index 3fb916c1..140d8176 100644 --- a/src/hapi/rrm/distrrm.py +++ b/src/hapi/rrm/distrrm.py @@ -119,7 +119,7 @@ def run_lumped_model(Model) -> SimulationResults: return results @staticmethod - def route_muskingum(Model): + def route_muskingum(Model, skip_hydraulic_cells: bool = False): """Route discharge between cells following the flow direction. Accumulates and routes upper-zone discharge (`quz`) using @@ -152,10 +152,12 @@ def route_muskingum(Model): - `Parameters` (numpy.ndarray): 3-D parameter array where indices 10 and 11 are Muskingum K and X. - `dt` (float): Time-step factor (`tfac`). - - `routing_method` (str): Routing method name (e.g., - `"Muskingum"`). - - `bankfull_depth` (numpy.ndarray): 2-D bankfull - depth array used for non-Muskingum methods. + - `bankfull_depth` (numpy.ndarray): 2-D bankfull depth array. Read only + when `skip_hydraulic_cells` is True. + + skip_hydraulic_cells: Skip cells with a positive `bankfull_depth`, because a + 1D hydraulic model routes them instead. The flood model's path; a plain + distributed run leaves it False and routes every cell. """ # # routing lake discharge with DS cell k & x and adding to cell Q # q_lake=Routing.muskingum_v(q_lake,q_lake[0],sp_pars[lakecell[0],lakecell[1],10],sp_pars[lakecell[0],lakecell[1],11],p2[0]) @@ -196,10 +198,13 @@ def route_muskingum(Model): not np.isnan(Model.flow_network.flow_acc_arr[x, y]) and Model.flow_network.flow_acc_arr[x, y] == acc_val[j] ): - if ( - Model.routing_method != "Muskingum" - and Model.bankfull_depth[x, y] > 0 - ): + if skip_hydraulic_cells and Model.bankfull_depth[x, y] > 0: + # A river cell a 1D hydraulic model will route instead. The + # caller says so explicitly; this used to be inferred from + # `routing_method != "Muskingum"`, which meant any catchment + # built with a non-Muskingum method dereferenced + # `bankfull_depth` -- None outside the flood model -- and + # crashed here. continue else: # for UZ diff --git a/src/hapi/run.py b/src/hapi/run.py index 610543b7..de14012b 100644 --- a/src/hapi/run.py +++ b/src/hapi/run.py @@ -199,7 +199,9 @@ def run_distributed(model: DistributedModel) -> SimulationResults: return results @staticmethod - def run_flood(model: FloodModel) -> SimulationResults: + def run_flood( + model: FloodModel, skip_hydraulic_cells: bool | None = None + ) -> SimulationResults: """Run the flood model. Runs the conceptual distributed hydrological model with @@ -208,9 +210,13 @@ def run_flood(model: FloodModel) -> SimulationResults: Args: model: The model to run. See :class:`FloodModel` for what it must carry. - - Returns: - SimulationResults: The run's output, also assigned to `model.results`. + skip_hydraulic_cells: Leave river cells (a positive `bankfull_depth`) unrouted + by the Muskingum pass, because the kinematic-wave model routes them instead. + `None`, the default, derives it from the catchment's own + `routing_method` -- `"Kinematic"` means yes, anything else no. Pass a bool to + override. The derivation used to live inside the routing loop as + `routing_method != "Muskingum"`, which ran for every distributed model and so + crashed `run_distributed` on a `bankfull_depth` of None. Raises: ValueError: If meteorological input arrays, parameter @@ -241,8 +247,13 @@ def run_flood(model: FloodModel) -> SimulationResults: if any(np.shape(arr)[1] != model.flow_network.cols for arr in geometry): raise ValueError("all input data should have the same number of columns") + if skip_hydraulic_cells is None: + skip_hydraulic_cells = model.routing_method == "Kinematic" + # run the model - results = Wrapper.run_muskingum(model) + results = Wrapper.run_muskingum( + model, skip_hydraulic_cells=skip_hydraulic_cells + ) logger.info("RRM has finished") # SV = SaintVenant() # SV.KinematicRaster(model) diff --git a/src/hapi/wrapper.py b/src/hapi/wrapper.py index e72ac006..29164f85 100644 --- a/src/hapi/wrapper.py +++ b/src/hapi/wrapper.py @@ -46,7 +46,10 @@ def __init__(self): @staticmethod def run_muskingum( - Model: DistributedModel, ll_temp=None, q_0=None + Model: DistributedModel, + ll_temp=None, + q_0=None, + skip_hydraulic_cells: bool = False, ) -> SimulationResults: """Run the distributed rainfall-runoff model with spatial routing. @@ -81,6 +84,12 @@ def run_muskingum( average temperature data. Defaults to None. q_0 (float, optional): Initial discharge in m3/s. Defaults to None. + skip_hydraulic_cells (bool, optional): Leave cells with a positive + `bankfull_depth` unrouted, because a 1D hydraulic model routes them. + The flood model's path. Defaults to False. + + Returns: + SimulationResults: The run's output, also assigned to `Model.results`. """ # run the rainfall runoff model separately results = distrrm.run_lumped_model(Model) @@ -88,7 +97,7 @@ def run_muskingum( # run the GIS part to rout from cell to another. It records # `RoutingKind.MUSKINGUM` on the results, which is what makes the outlet-cell # shortcut in `extract_discharge` valid for them. - distrrm.route_muskingum(Model) + distrrm.route_muskingum(Model, skip_hydraulic_cells=skip_hydraulic_cells) return results @staticmethod diff --git a/tests/rrm/catchment/test_config.py b/tests/rrm/catchment/test_config.py index e73a2459..4cb6565e 100644 --- a/tests/rrm/catchment/test_config.py +++ b/tests/rrm/catchment/test_config.py @@ -1226,12 +1226,14 @@ def test_any_casing_is_stored_canonically(self, given, stored): ) def test_kinematic_is_accepted_for_the_flood_model(self): - """Test that the flood model's routing method is still a legal value. + """Test that the kinematic-wave routing method stays a legal value. Test scenario: - `Run.run_flood` relies on the same `!= "Muskingum"` comparison to skip cells - with a real `bankfull_depth`, so "Kinematic" is a working value and must not be - rejected by the new validation. + "Kinematic" names the routing the flood model applies to river cells. It is read + by `Run.run_flood`, which uses it to decide whether the Muskingum pass leaves + those cells to the hydraulic model -- so it carries meaning and must be accepted. + What changed is only where it is read: no longer inside the routing loop, where + it ran for every distributed model. """ model = Catchment( "coello", "2009-01-01", "2009-01-10", routing_method="Kinematic" diff --git a/tests/rrm/catchment/test_run_results_coupling.py b/tests/rrm/catchment/test_run_results_coupling.py index 203366ec..e075bd12 100644 --- a/tests/rrm/catchment/test_run_results_coupling.py +++ b/tests/rrm/catchment/test_run_results_coupling.py @@ -383,3 +383,62 @@ def test_the_flood_model_names_the_river_geometry_it_lacks( assert "bankfull_depth" in str(exc_info.value), ( f"the error should name the missing rasters, got: {exc_info.value}" ) + + +class TestRoutingIsChosenByTheEntryPoint: + """`routing_method` records intent; the entry point is what routes.""" + + @pytest.mark.parametrize("declared", ["Muskingum", "MAXBAS"]) + def test_run_distributed_routes_with_muskingum_whatever_the_field_says( + self, + coello_fixtures: dict, + coello_dist_parameters_muskingum: str, + declared: str, + ): + """Test that `routing_method` does not change what `run_distributed` does. + + Test scenario: + The field used to be compared inside the routing loop, so a catchment built with + anything but `"Muskingum"` reached `bankfull_depth[x, y]` -- None outside the + flood model -- and died with `TypeError: 'NoneType' object is not subscriptable` + partway through routing. Skipping river cells is now an explicit argument to the + flood entry point, so the declared method cannot reach the loop at all. + + Args: + declared: The routing method the catchment is built with. + """ + model = _build("coello", coello_dist_parameters_muskingum, **coello_fixtures) + model.routing_method = declared + + results = Run.run_distributed(model) + + assert results.routing is RoutingKind.MUSKINGUM, ( + f"run_distributed must route with Muskingum whatever routing_method says; " + f"declared {declared!r}, got {results.routing}" + ) + assert results.q_total is not None, "every cell must have been routed" + + def test_a_kinematic_catchment_still_runs_distributed_without_crashing( + self, coello_fixtures: dict, coello_dist_parameters_muskingum: str + ): + """Test that declaring Kinematic no longer breaks a plain distributed run. + + Test scenario: + `"Kinematic"` is a real routing method -- the wave model the flood path applies + to river cells -- and stays accepted. But the routing loop used to compare + against it directly, so a catchment declaring it and calling `run_distributed` + reached `bankfull_depth[x, y]`, which is None outside the flood model, and died + with `TypeError: 'NoneType' object is not subscriptable` partway through routing. + The skip is read by `run_flood` now, so `run_distributed` routes every cell. + """ + model = _build("coello", coello_dist_parameters_muskingum, **coello_fixtures) + model.routing_method = "Kinematic" + + results = Run.run_distributed(model) + + assert results.routing is RoutingKind.MUSKINGUM, ( + "run_distributed routes with Muskingum regardless of the declared method" + ) + assert results.q_total is not None, ( + "every cell must be routed; the flood skip belongs to run_flood" + ) diff --git a/tests/rrm/catchment/test_run_validation.py b/tests/rrm/catchment/test_run_validation.py index 6f3f0d9e..62363bfe 100644 --- a/tests/rrm/catchment/test_run_validation.py +++ b/tests/rrm/catchment/test_run_validation.py @@ -38,7 +38,7 @@ def spied_wrapper(monkeypatch) -> dict: calls: dict = {} def _make(name: str): - def _spy(*args): + def _spy(*args, **kwargs): calls[name] = args return _spy @@ -296,3 +296,66 @@ def test_rejects_parameters_off_the_catchment_grid( assert "run_maxbas_with_lake" not in spied_wrapper, ( "the wrapper must not run on mis-shaped parameters" ) + + +class TestFloodModelHonoursKinematic: + """`run_flood` reads `routing_method` to decide whether Muskingum skips the river cells.""" + + @pytest.mark.parametrize( + "declared, expected_skip", + [("Kinematic", True), ("Muskingum", False)], + ) + def test_the_skip_is_derived_from_the_declared_routing( + self, coello_loaded: Catchment, monkeypatch, declared: str, expected_skip: bool + ): + """Test that the kinematic-wave declaration reaches the routing as a real argument. + + Test scenario: + `"Kinematic"` means the wave model routes the river cells, so the Muskingum pass + must leave them alone. That used to be a `routing_method != "Muskingum"` compare + *inside* the routing loop, which is why it also fired on plain distributed runs. + It is now read once, by this entry point, and passed down explicitly. + + Args: + declared: The routing method the catchment declares. + expected_skip: Whether the wrapper should be told to skip hydraulic cells. + """ + _load_flat_river_geometry(coello_loaded) + coello_loaded.routing_method = declared + seen: dict = {} + + def _spy(model, ll_temp=None, q_0=None, skip_hydraulic_cells=False): + seen["skip"] = skip_hydraulic_cells + + monkeypatch.setattr(run_module.Wrapper, "run_muskingum", staticmethod(_spy)) + + Run.run_flood(coello_loaded) + + assert seen["skip"] is expected_skip, ( + f"routing_method={declared!r} should give skip_hydraulic_cells=" + f"{expected_skip}, got {seen['skip']}" + ) + + def test_an_explicit_argument_overrides_the_declaration( + self, coello_loaded: Catchment, monkeypatch + ): + """Test that a caller can ask for the skip without declaring Kinematic. + + Test scenario: + Deriving from `routing_method` keeps the historical spelling working, but the + request is a run-time choice, so it stays expressible directly. + """ + _load_flat_river_geometry(coello_loaded) + coello_loaded.routing_method = "Muskingum" + seen: dict = {} + + def _spy(model, ll_temp=None, q_0=None, skip_hydraulic_cells=False): + seen["skip"] = skip_hydraulic_cells + + monkeypatch.setattr(run_module.Wrapper, "run_muskingum", staticmethod(_spy)) + + Run.run_flood(coello_loaded, skip_hydraulic_cells=True) + + assert seen["skip"] is True, ( + "an explicit argument must win over the declaration" + ) From ec8726cd45e2649f0ffad1eccc0c1777ae797a23 Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Wed, 2 Sep 2026 00:26:39 +0200 Subject: [PATCH 09/54] feat(examples): restore a runnable flood-model script, and warn what Kinematic costs `tests/FloodModel.py` left this repository in `733957be` ("move flood model to serapis", 2024-01-03) and still sits in Serapis, byte-identical apart from one `os.chdir` path. It cannot run against this package any more: it points at `F:/02Case-studies/`, imports the pre-rename `Hapi` package, passes a module where `read_lumped_model` wants a class, and calls five readers that no longer exist -- `read_flow_acc`, `read_flow_dir`, `read_rainfall`, `read_temperature` and `read_et`, all of which moved into `FlowNetwork` and `MeteoInputs`. Ported to `examples/hydrological-model/coello/run/coello-flood-model-run.py` against the five river-geometry rasters already in the example data (dem4000, bankfulldepth, river_width, channel_roughness, floodplain_roughness). It runs. Running it also showed what the Kinematic path now does. Declaring `routing_method="Kinematic"` tells the Muskingum pass to leave the river cells to a 1D hydraulic model -- the handoff the flood model was designed around -- but that model went to Serapis in the same commit, so nothing here picks them up. On the Coello example that is 21 of 89 catchment cells and 92% of the discharge, absent from the results with nothing said. It was unreachable until the previous commit, because the comparison crashed on a `bankfull_depth` of None; making it reachable turned a crash into a silent hole. So the derived case warns, naming the counts and both ways out. An explicit `skip_hydraulic_cells=True` is a statement that something downstream takes the cells and stays quiet -- that is the supported workflow, and it should not nag. The counts are taken inside the flow-accumulation mask. The bankfull-depth raster carries values outside the catchment too, and counting the raster alone had the message claiming more river cells than the catchment has. --- .../coello/run/coello-flood-model-run.py | 105 ++++++++++++++++++ src/hapi/run.py | 62 +++++++++-- tests/rrm/catchment/test_run_validation.py | 62 +++++++++++ 3 files changed, 220 insertions(+), 9 deletions(-) create mode 100644 examples/hydrological-model/coello/run/coello-flood-model-run.py diff --git a/examples/hydrological-model/coello/run/coello-flood-model-run.py b/examples/hydrological-model/coello/run/coello-flood-model-run.py new file mode 100644 index 00000000..eb2eac0a --- /dev/null +++ b/examples/hydrological-model/coello/run/coello-flood-model-run.py @@ -0,0 +1,105 @@ +"""Distributed model on the flood-model path, with kinematic-wave routing declared. + +Ported from `tests/FloodModel.py`, which left this repository in commit `733957be` +("move flood model to serapis", 2024-01-03) and still sits unchanged in Serapis. That +script could not run here any more: it pointed at `F:/02Case-studies/`, imported the +pre-rename `Hapi` package, and called five readers that no longer exist -- +`read_flow_acc`, `read_flow_dir`, `read_rainfall`, `read_temperature` and `read_et`, +all of which moved into `FlowNetwork` and `MeteoInputs`. + +What the flood path does, and does not, do: + +`Run.run_flood` validates the four river-geometry rasters against the catchment grid and +then runs the ordinary Muskingum distributed model. Declaring `routing_method="Kinematic"` +tells it that a 1D hydraulic model routes the river cells, so the Muskingum pass leaves +them alone -- that is the handoff the original design intended, and the hydraulic half +(`SaintVenant.KinematicRaster`) went to Serapis in the same commit. Nothing in Hapi picks +those cells up, so **with `"Kinematic"` the river cells carry no routed discharge**. Run it +with `"Muskingum"` -- the default below -- to route every cell here, and use `"Kinematic"` +only when Serapis will take the river cells from you. + +The paths are written from the repo root, so run this from there: +`python examples/hydrological-model/coello/run/coello-flood-model-run.py`. +""" + +from __future__ import annotations + +import numpy as np + +from hapi.catchment import Catchment +from hapi.inputs import FlowNetwork, MeteoInputs +from hapi.rrm.hbv_bergestrom92 import HBVBergestrom92 +from hapi.run import Run + +# %% Paths +DATA = "examples/hydrological-model/data/distributed_model" +GIS = f"{DATA}/GIS" + +PREC_PATH = f"{DATA}/prec" +EVAP_PATH = f"{DATA}/evap" +TEMP_PATH = f"{DATA}/temp" +FLOW_ACC_PATH = f"{GIS}/acc4000.tif" +FLOW_DIR_PATH = f"{GIS}/fd4000.tif" +PARAMETERS_PATH = f"{DATA}/Parameter set-Avg" + +# %% Flood-model rasters +DEM_FILE = f"{GIS}/dem4000.tif" +BANKFULL_DEPTH_FILE = f"{GIS}/bankfulldepth.tif" +RIVER_WIDTH_FILE = f"{GIS}/river_width.tif" +RIVER_ROUGHNESS_FILE = f"{GIS}/channel_roughness.tif" +FLOODPLAIN_ROUGHNESS_FILE = f"{GIS}/floodplain_roughness.tif" + +# "Muskingum" routes every cell here. Use "Kinematic" only when a hydraulic model +# downstream will route the river cells -- see the module docstring. +ROUTING_METHOD = "Muskingum" + +# %% Model configuration +AREA = 1530 +INITIAL_COND = [0, 5, 5, 5, 0] +SNOW = False +START = "2009-01-01" +END = "2009-01-10" + +# %% Build the model +Coello = Catchment( + "Coello", + START, + END, + spatial_resolution="Distributed", + routing_method=ROUTING_METHOD, +) + +Coello.meteo = MeteoInputs.from_rasters( + PREC_PATH, + TEMP_PATH, + EVAP_PATH, + start=START, + end=END, + regex_string=r"\d{4}.\d{2}.\d{2}", + date=True, + file_name_data_fmt="%Y.%m.%d", +) +Coello.flow_network = FlowNetwork.from_rasters(FLOW_ACC_PATH, FLOW_DIR_PATH) +Coello.read_river_geometry( + DEM_FILE, + BANKFULL_DEPTH_FILE, + RIVER_WIDTH_FILE, + RIVER_ROUGHNESS_FILE, + FLOODPLAIN_ROUGHNESS_FILE, +) +Coello.read_parameters(PARAMETERS_PATH, SNOW) +Coello.read_lumped_model(HBVBergestrom92, AREA, INITIAL_COND) + +# %% Run the flood model +results = Run.run_flood(Coello) + +# The domain is the flow-accumulation mask, not the whole grid: `q_total` is allocated +# with zeros everywhere, so counting its non-NaN cells would just report rows x cols. +inside = ~np.isnan(Coello.flow_network.flow_acc_arr) +river_cells = int(np.count_nonzero((np.nan_to_num(Coello.bankfull_depth) > 0) & inside)) + +print(f"routing : {results.routing.value}") +print(f"q_total : {results.q_total.shape}") +print(f"catchment cells : {int(np.count_nonzero(inside))}") +print(f"river cells : {river_cells} (left to a hydraulic model under 'Kinematic')") +print(f"basin-wide q_total : {float(np.nansum(results.q_total)):.1f}") diff --git a/src/hapi/run.py b/src/hapi/run.py index de14012b..5c9bb808 100644 --- a/src/hapi/run.py +++ b/src/hapi/run.py @@ -14,6 +14,7 @@ from __future__ import annotations +import warnings from collections.abc import Callable from typing import TYPE_CHECKING, Any @@ -90,6 +91,42 @@ def _check_lake_meteo(model: DistributedModel, lake: LakeType) -> None: ) +def _warn_about_the_unrouted_river_cells(model: FloodModel) -> None: + """Warn that the river cells will be left unrouted, and by how much. + + Skipping them is the handoff the flood model was designed around: a 1D hydraulic model + (`SaintVenant.KinematicRaster`) routes them instead. That half left this package in + commit `733957be` and now lives in Serapis, so nothing here picks those cells up -- on + the Coello example that is 21 of the 89 catchment cells, carrying 92% of the discharge + because the river cells are the high-accumulation ones. A caller who has + Serapis downstream wants exactly this; a caller who set `routing_method="Kinematic"` + without one gets a number that is not a hydrograph, and nothing used to say so. + + Only the *derived* case warns. Passing `skip_hydraulic_cells=True` is an explicit + statement that something downstream takes the river cells, and is left quiet. + + Args: + model: The model about to run, whose `bankfull_depth` marks the river cells. + """ + # Only cells inside the catchment are ever routed, so only those can be skipped. The + # bankfull-depth raster carries values outside the domain too, and counting those made + # the message claim more river cells than the catchment has. + inside = ~np.isnan(model.flow_network.flow_acc_arr) + river_cells = int( + np.count_nonzero((np.nan_to_num(model.bankfull_depth) > 0) & inside) + ) + domain = int(np.count_nonzero(inside)) + warnings.warn( + f"routing_method='Kinematic' leaves the {river_cells} river cells of {domain} " + "unrouted, for a 1D hydraulic model to route instead -- but that model is not part " + "of Hapi, so their discharge is simply absent from the results. Pass " + "skip_hydraulic_cells=True to say a downstream model takes them and silence this, " + "or use routing_method='Muskingum' to route every cell here.", + UserWarning, + stacklevel=3, + ) + + def _validate_distributed(model: DistributedModel, check_flow_direction: bool) -> None: """Run the checks every distributed entry point makes before the wrapper. @@ -213,10 +250,14 @@ def run_flood( skip_hydraulic_cells: Leave river cells (a positive `bankfull_depth`) unrouted by the Muskingum pass, because the kinematic-wave model routes them instead. `None`, the default, derives it from the catchment's own - `routing_method` -- `"Kinematic"` means yes, anything else no. Pass a bool to - override. The derivation used to live inside the routing loop as - `routing_method != "Muskingum"`, which ran for every distributed model and so - crashed `run_distributed` on a `bankfull_depth` of None. + `routing_method` -- `"Kinematic"` means yes, anything else no -- and warns, + because the hydraulic model that was supposed to take those cells is not part + of Hapi. Pass `True` to state that something downstream takes them and + silence the warning, or `False` to route every cell here. + + Warns: + UserWarning: The skip was derived from `routing_method="Kinematic"`, so the river + cells are left unrouted and their discharge is absent from the results. Raises: ValueError: If meteorological input arrays, parameter @@ -247,13 +288,16 @@ def run_flood( if any(np.shape(arr)[1] != model.flow_network.cols for arr in geometry): raise ValueError("all input data should have the same number of columns") - if skip_hydraulic_cells is None: - skip_hydraulic_cells = model.routing_method == "Kinematic" + derived = skip_hydraulic_cells is None + skip = ( + model.routing_method == "Kinematic" if derived else bool(skip_hydraulic_cells) + ) + + if skip and derived: + _warn_about_the_unrouted_river_cells(model) # run the model - results = Wrapper.run_muskingum( - model, skip_hydraulic_cells=skip_hydraulic_cells - ) + results = Wrapper.run_muskingum(model, skip_hydraulic_cells=skip) logger.info("RRM has finished") # SV = SaintVenant() # SV.KinematicRaster(model) diff --git a/tests/rrm/catchment/test_run_validation.py b/tests/rrm/catchment/test_run_validation.py index 62363bfe..242a069d 100644 --- a/tests/rrm/catchment/test_run_validation.py +++ b/tests/rrm/catchment/test_run_validation.py @@ -8,6 +8,8 @@ from __future__ import annotations +import warnings + import numpy as np import pytest @@ -359,3 +361,63 @@ def _spy(model, ll_temp=None, q_0=None, skip_hydraulic_cells=False): assert seen["skip"] is True, ( "an explicit argument must win over the declaration" ) + + def test_a_derived_kinematic_skip_warns_that_the_river_cells_go_unrouted( + self, coello_loaded: Catchment, monkeypatch + ): + """Test that inferring the skip from `routing_method` says what it costs. + + Test scenario: + Skipping the river cells is the handoff the flood model was built around -- a 1D + hydraulic model routes them instead -- but that model left this package for + Serapis, so nothing here picks them up and their discharge is simply absent. + Someone who set `routing_method="Kinematic"` without a downstream model gets a + number that is not a hydrograph, so the derived case has to say so. + """ + _load_flat_river_geometry(coello_loaded) + coello_loaded.routing_method = "Kinematic" + monkeypatch.setattr( + run_module.Wrapper, "run_muskingum", staticmethod(lambda *a, **k: None) + ) + + with pytest.warns(UserWarning, match="unrouted") as record: + Run.run_flood(coello_loaded) + + message = str(record[0].message) + assert "skip_hydraulic_cells=True" in message, ( + f"the warning should name the way to silence it, got: {message}" + ) + # The geometry helper marks every cell as a river cell, and the Coello grid has 89 + # inside the flow-accumulation mask -- so the count must be the domain, not rows x + # cols, which is what counting the raster alone reported. + inside = int( + np.count_nonzero(~np.isnan(coello_loaded.flow_network.flow_acc_arr)) + ) + assert f"{inside} river cells of {inside}" in message, ( + f"the counts must be taken inside the catchment, got: {message}" + ) + + def test_an_explicit_skip_does_not_warn( + self, coello_loaded: Catchment, monkeypatch + ): + """Test that asking for the skip outright is left quiet. + + Test scenario: + `skip_hydraulic_cells=True` is a statement that something downstream takes the + river cells. That is the supported workflow, so it must not nag every run; the + warning exists for the case nobody chose. + """ + _load_flat_river_geometry(coello_loaded) + coello_loaded.routing_method = "Kinematic" + monkeypatch.setattr( + run_module.Wrapper, "run_muskingum", staticmethod(lambda *a, **k: None) + ) + + with warnings.catch_warnings(record=True) as caught: + warnings.simplefilter("always") + Run.run_flood(coello_loaded, skip_hydraulic_cells=True) + + unrouted = [w for w in caught if "unrouted" in str(w.message)] + assert not unrouted, ( + f"an explicit skip must not warn, got: {[str(w.message) for w in unrouted]}" + ) From a1481d28abe2e4867f1ea8eb6db8e3367271f4e9 Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Sun, 6 Sep 2026 18:19:52 +0200 Subject: [PATCH 10/54] refactor(runs)!: split the builder from the finished model `Catchment` was both things at once. It is a builder -- constructed empty, filled by `read_*` calls in any order, some of which a given run never needs -- so every input on it is `X | None`, and that is honest. The run layer needs the opposite: a catchment that is finished. Conflating the two had three costs, and only the first was cosmetic: 1. The engines dereferenced `X | None` on every line, which is why `run`, `wrapper` and `distrrm` were excused from mypy. 2. "Has this been validated?" was answered by remembering which entry point you came through. `Calibration` went straight to `Wrapper` and so skipped every check `Run` performed -- on the one path that rebuilds the parameter array thousands of times. 3. The engines wrote results back onto the object they read, so a half-finished run and a finished one looked alike. `hapi.runs` adds `DistributedRun` and `LumpedRun`: frozen, non-optional, and reachable only through `from_model`, which is now the single validation seam -- constructing one *is* the validation. The engines take these, so the check is enforced by the signatures rather than by discipline: `Calibration` cannot bypass it any more because there is nothing else to pass. `hapi.protocols` keeps the builder side honest as `CatchmentLike`, whose fields really are optional, and `Catchment` satisfies it cleanly -- verified with mypy, where it previously failed on five `| None` conflicts. Results now only ever come back as return values. `Wrapper.run_lumped` puts the lumped total in `results.q_total` -- which is what `q_total` means -- and the entry point indexes it by the period, so `results` is the sole write to the model. `RiverGeometry` (`hapi.inputs`) holds the five hydraulic rasters that were five loose attributes assigned by a loop that checked nothing: not that they shared a shape, not that they covered the catchment. Both are settled now, the first where the file names are still in hand. Taking `hapi.wrapper` off the suppression list surfaced four real defects it had been hiding, all fixed here: `Lake.Qlake` and `Lake.QlakeR` were created by assignment from inside the wrapper and declared nowhere; `Lake.Parameters` and `Lake.OutflowCell` were indexed straight while typed optional; and `ConceptualModelSetup.model` was optional though `read_lumped_model` always builds one. `DistributedRun` also exposes `parameter_cube` and `routing_table`, so the engines index a real array and a real dict instead of a union. `run`, `wrapper` and `distrrm` are off the mypy suppression list; the whole package type-checks clean. 588 tests pass, 18 of them new and covering the seam itself. BREAKING CHANGE: `Wrapper.*` and `DistributedRRM.*` take a `DistributedRun` or `LumpedRun`, not a `Catchment` -- build one with `DistributedRun.from_model(model)`. `DistributedRRM.route_muskingum` / `route_maxbas` / `route_maxbas_by_path_length` also take the `SimulationResults` to fill. `Catchment.DEM`, `.bankfull_depth`, `.river_width`, `.river_roughness` and `.flood_plain_roughness` are gone; read them off `model.river_geometry`. `Wrapper.run_lumped` no longer sets `Qsim` -- it fills `results.q_total`, and `Run.run_lumped` puts the frame on the model. --- .../coello/run/coello-flood-model-run.py | 8 +- pyproject.toml | 10 +- src/hapi/calibration.py | 20 +- src/hapi/catchment.py | 33 ++- src/hapi/conceptual.py | 7 +- src/hapi/inputs.py | 127 ++++++++ src/hapi/protocols.py | 117 +++----- src/hapi/rrm/distrrm.py | 269 +++++++---------- src/hapi/run.py | 182 ++++-------- src/hapi/runs.py | 277 ++++++++++++++++++ src/hapi/wrapper.py | 189 +++++++----- tests/rrm/catchment/test_fw1_output_fields.py | 5 +- .../catchment/test_maxbas_routing_variants.py | 28 +- tests/rrm/catchment/test_run_narrowing.py | 249 ++++++++++++++++ .../catchment/test_run_results_coupling.py | 11 +- tests/rrm/catchment/test_run_validation.py | 50 ++-- tests/rrm/catchment/test_wrapper_with_lake.py | 33 ++- 17 files changed, 1101 insertions(+), 514 deletions(-) create mode 100644 src/hapi/runs.py create mode 100644 tests/rrm/catchment/test_run_narrowing.py diff --git a/examples/hydrological-model/coello/run/coello-flood-model-run.py b/examples/hydrological-model/coello/run/coello-flood-model-run.py index eb2eac0a..4be07431 100644 --- a/examples/hydrological-model/coello/run/coello-flood-model-run.py +++ b/examples/hydrological-model/coello/run/coello-flood-model-run.py @@ -96,10 +96,14 @@ # The domain is the flow-accumulation mask, not the whole grid: `q_total` is allocated # with zeros everywhere, so counting its non-NaN cells would just report rows x cols. inside = ~np.isnan(Coello.flow_network.flow_acc_arr) -river_cells = int(np.count_nonzero((np.nan_to_num(Coello.bankfull_depth) > 0) & inside)) +river_cells = int( + np.count_nonzero((np.nan_to_num(Coello.river_geometry.bankfull_depth) > 0) & inside) +) print(f"routing : {results.routing.value}") print(f"q_total : {results.q_total.shape}") print(f"catchment cells : {int(np.count_nonzero(inside))}") -print(f"river cells : {river_cells} (left to a hydraulic model under 'Kinematic')") +print( + f"river cells : {river_cells} (left to a hydraulic model under 'Kinematic')" +) print(f"basin-wide q_total : {float(np.nansum(results.q_total)):.1f}") diff --git a/pyproject.toml b/pyproject.toml index d06f4d53..71850d4e 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -188,11 +188,11 @@ exclude = [ module = [ "hapi.catchment", "hapi.calibration", - "hapi.wrapper", ] -# These modules are built by successive read_*() calls: attributes are typed -# X | None and populated before use, which mypy cannot verify. Removing the -# suppressions needs assert-based narrowing helpers first (~257 errors). +# These two are builders: attributes are typed X | None and populated by successive +# read_*() calls, which mypy cannot verify. The run layer no longer needs the excuse -- +# `hapi.runs` narrows a builder into a validated, non-optional run, so `run`, `wrapper` +# and `distrrm` type-check clean. What is left here is the builder side itself. disable_error_code = [ "union-attr", "attr-defined", @@ -262,7 +262,7 @@ cmd = [ "pytest", "--doctest-modules", "-p", "no:cacheprovider", "--no-cov", "src/hapi/config.py", "src/hapi/catchment.py", "src/hapi/conceptual.py", "src/hapi/inputs.py", "src/hapi/period.py", - "src/hapi/results.py", + "src/hapi/results.py", "src/hapi/runs.py", "src/hapi/routing.py", "src/hapi/run.py", ] diff --git a/src/hapi/calibration.py b/src/hapi/calibration.py index c0d4bf45..9093c227 100644 --- a/src/hapi/calibration.py +++ b/src/hapi/calibration.py @@ -17,6 +17,7 @@ from hapi.catchment import Catchment from hapi.conceptual import ParameterSet +from hapi.runs import DistributedRun, LumpedRun from hapi.wrapper import Wrapper ROWS_MISMATCH_ERROR = "all input data should have the same number of rows" @@ -356,8 +357,10 @@ def opt_fun(par): # rule, so a distribution function producing the wrong width fails here # rather than as an index error inside the per-cell loop. self.parameters = self._parameter_set(spatial_var_fun.Par3d) - # run the model - Wrapper.run_muskingum(self) + # Narrowing is the validation: every trial vector is checked against the grid + # and the period here, on the one path that used to skip every check `Run` + # made by going straight to the wrapper. + self.results = Wrapper.run_muskingum(DistributedRun.from_model(self)) # calculate performance of the model try: error = self.objective_function( @@ -503,8 +506,10 @@ def opt_fun(par): ) # , kub=spatial_var_fun.Kub, klb=spatial_var_fun.Klb, Maskingum=spatial_var_fun.Maskingum # Re-checked per trial -- see run_calibration. self.parameters = self._parameter_set(spatial_var_fun.Par3d) - # run the model - Wrapper.run_maxbas(self) + # See run_calibration: narrowing validates each trial vector. + self.results = Wrapper.run_maxbas( + DistributedRun.from_model(self, needs_flow_direction=False) + ) # calculate performance of the model try: error = self.objective_function( @@ -637,8 +642,11 @@ def opt_fun(par): try: # parameters. Checked against (snow, maxbas) as it arrives. self.parameters = self._parameter_set(par) - # run the model - Wrapper.run_lumped(self, route, routing_fn) + # See run_calibration: narrowing validates each trial vector. + self.results = Wrapper.run_lumped( + LumpedRun.from_model(self), route, routing_fn + ) + self.Qsim = self.results.q_total # calculate performance of the model try: error = self.objective_function( diff --git a/src/hapi/catchment.py b/src/hapi/catchment.py index 4eb4d8b3..74d8b9d8 100644 --- a/src/hapi/catchment.py +++ b/src/hapi/catchment.py @@ -41,6 +41,7 @@ METEO_VARIABLES, FlowNetwork, MeteoInputs, + RiverGeometry, _warn_if_no_sentinel, read_rasters, ) @@ -328,11 +329,9 @@ def __init__( #: :class:`~hapi.inputs.FlowNetwork` built by its loader. self.flow_network: FlowNetwork | None = None self.flow_path_length_arr: np.ndarray | None = None - self.DEM: np.ndarray | None = None - self.bankfull_depth: np.ndarray | None = None - self.river_width: np.ndarray | None = None - self.river_roughness: np.ndarray | None = None - self.flood_plain_roughness: np.ndarray | None = None + #: The five hydraulic rasters the flood model reads, once `read_river_geometry` has + #: run. Absent-or-complete: they are checked against each other as they are read. + self.river_geometry: RiverGeometry | None = None #: Everything one run produced, replaced wholesale by the next run. The seven #: result arrays below are read-only properties forwarding to it, so `model.results.q_total` #: still reads as it always did while the run layer owns the arrays. `None` until @@ -615,15 +614,16 @@ def read_river_geometry( floodplain_roughness_file (str): Path to the floodplain roughness raster file. """ - for name, fpath in [ - ("DEM", dem_file), - ("bankfull_depth", bankfull_depth_file), - ("river_width", river_width_file), - ("river_roughness", river_roughness_file), - ("flood_plain_roughness", floodplain_roughness_file), - ]: - ds = Dataset.read_file(fpath) - setattr(self, name, ds.read_array(band=0)) + # One object rather than five loose arrays: `RiverGeometry` checks they share a grid + # as it reads them, where the file names are still in hand and the error can name the + # odd one out. The loop this replaces checked nothing. + self.river_geometry = RiverGeometry.from_rasters( + dem_file, + bankfull_depth_file, + river_width_file, + river_roughness_file, + floodplain_roughness_file, + ) def read_parameters(self, path: str, snow: bool = False, maxbas: bool = False): """Read model parameter rasters or a CSV parameter file. @@ -1684,6 +1684,11 @@ def __init__( self.LakeArea: float | None = None self.InitialCond: list | None = None self.StageDischargeCurve: np.ndarray | None = None + #: The lake's own simulated outflow, and that series routed to the outflow cell. + #: Filled by the lake-aware wrapper entry points, which used to create them by + #: assignment -- so they existed only after a run and nothing said they were coming. + self.Qlake: np.ndarray | None = None + self.QlakeR: np.ndarray | None = None def read_meteo_data(self, path: str, fmt: str): """Read meteorological data for the lake simulation. diff --git a/src/hapi/conceptual.py b/src/hapi/conceptual.py index 2e187bd8..a9c78e11 100644 --- a/src/hapi/conceptual.py +++ b/src/hapi/conceptual.py @@ -233,14 +233,17 @@ class ConceptualModelSetup: Examples: ```python >>> from hapi.conceptual import ConceptualModelSetup - >>> setup = ConceptualModelSetup(None, 1530.0, [0, 10, 10, 10, 0], q_init=5.0) + >>> from hapi.rrm.hbv_bergestrom92 import HBVBergestrom92 + >>> setup = ConceptualModelSetup( + ... HBVBergestrom92(), 1530.0, [0, 10, 10, 10, 0], q_init=5.0 + ... ) >>> setup.area, setup.q_init (1530.0, 5.0) ``` """ - model: BaseConceptualModel | None + model: BaseConceptualModel area: float | int initial_cond: list q_init: float | None = None diff --git a/src/hapi/inputs.py b/src/hapi/inputs.py index 1c950a47..7898073c 100644 --- a/src/hapi/inputs.py +++ b/src/hapi/inputs.py @@ -1579,6 +1579,133 @@ def _calendar(nc: NetCDF) -> pd.DatetimeIndex | None: ] +#: The five rasters the flood model reads, in the order `read_river_geometry` takes them. +RIVER_GEOMETRY_RASTERS = ( + "dem", + "bankfull_depth", + "river_width", + "river_roughness", + "flood_plain_roughness", +) + + +@dataclass(frozen=True) +class RiverGeometry: + """The hydraulic rasters the flood model reads, held together and checked as a set. + + Five arrays that must describe one grid, previously five loose attributes on + :class:`~hapi.catchment.Catchment` assigned by a single loop that checked nothing -- not + that they shared a shape, and not that they covered the catchment. Two unrelated places + then dereferenced them: the flood entry point, and the Muskingum routing loop, which read + `bankfull_depth[x, y]` to decide whether a cell belongs to a hydraulic model. + + Held together, the invariant is checked once, where the file names are still in hand and + the error can say which raster is the odd one out. + + Attributes: + dem: `(rows, cols)` elevation. + bankfull_depth: `(rows, cols)` bankfull depth. A positive value marks a river cell. + river_width: `(rows, cols)` channel width. + river_roughness: `(rows, cols)` channel roughness. + flood_plain_roughness: `(rows, cols)` floodplain roughness. + + Examples: + - The five must share a shape: + ```python + >>> import numpy as np + >>> from hapi.inputs import RiverGeometry + >>> flat = np.ones((3, 4)) + >>> RiverGeometry(flat, flat, flat, flat, flat).shape + (3, 4) + + ``` + - A raster of the wrong size is named, not silently carried: + ```python + >>> import numpy as np + >>> from hapi.inputs import RiverGeometry + >>> flat = np.ones((3, 4)) + >>> RiverGeometry(flat, flat, np.ones((2, 4)), flat, flat) + Traceback (most recent call last): + ... + ValueError: the river geometry rasters must share one grid, but river_width is (2, 4) and dem is (3, 4) + + ``` + """ + + dem: np.ndarray + bankfull_depth: np.ndarray + river_width: np.ndarray + river_roughness: np.ndarray + flood_plain_roughness: np.ndarray + + def __post_init__(self): + """Check the five rasters describe one grid. + + Raises: + ValueError: A raster is a different shape from the DEM, so a cell index would mean + a different place in each. + """ + expected = np.shape(self.dem) + for name in RIVER_GEOMETRY_RASTERS[1:]: + shape = np.shape(getattr(self, name)) + if shape != expected: + raise ValueError( + f"the river geometry rasters must share one grid, but {name} is {shape} " + f"and dem is {expected}" + ) + + @property + def shape(self) -> tuple[int, int]: + """tuple[int, int]: The `(rows, cols)` grid all five share.""" + return np.shape(self.dem) + + def covers(self, rows: int, cols: int) -> bool: + """Report whether the geometry sits on a given grid. + + Args: + rows: Number of rows to compare against. + cols: Number of columns. + + Returns: + bool: True when the geometry's grid is exactly `(rows, cols)`. + """ + return self.shape == (rows, cols) + + @classmethod + def from_rasters( + cls, + dem_file: str, + bankfull_depth_file: str, + river_width_file: str, + river_roughness_file: str, + floodplain_roughness_file: str, + ) -> RiverGeometry: + """Read the five rasters and check they agree before returning them. + + Args: + dem_file: Path to the DEM raster. + bankfull_depth_file: Path to the bankfull-depth raster. + river_width_file: Path to the river-width raster. + river_roughness_file: Path to the channel-roughness raster. + floodplain_roughness_file: Path to the floodplain-roughness raster. + + Returns: + RiverGeometry: The five arrays, sharing one grid. + + Raises: + FileNotFoundError: A path does not exist. + ValueError: The rasters do not share a grid. + """ + paths = ( + dem_file, + bankfull_depth_file, + river_width_file, + river_roughness_file, + floodplain_roughness_file, + ) + return cls(*(Dataset.read_file(path).read_array(band=0) for path in paths)) + + class Inputs: """Rainfall-runoff inputs preparation for distributed hydrological models. diff --git a/src/hapi/protocols.py b/src/hapi/protocols.py index 4f12f253..0e7088cb 100644 --- a/src/hapi/protocols.py +++ b/src/hapi/protocols.py @@ -1,18 +1,22 @@ -"""What the run layer requires of the model it is handed. - -The entry points in :mod:`hapi.run` and the wiring in :mod:`hapi.wrapper` used to declare -their argument as :class:`~hapi.catchment.Catchment` -- a class of 40-odd attributes, of which -any one run touches a dozen. That named the wrong thing: it over-stated the requirement, and -it pointed the dependency at a concrete class, so the run layer could not be reasoned about -without the class it runs. - -These protocols state the requirement instead. `Catchment` satisfies them structurally without -inheriting anything, so neither `hapi.run` nor `hapi.wrapper` imports it at runtime, and any -other object carrying the same attributes runs too. - -They live in their own module rather than in `hapi.run` because `hapi.run` imports -`hapi.wrapper`, and `hapi.wrapper` needs the same protocols -- a shared home is what keeps -that from being a cycle. +"""What the run layer accepts from a caller, before it has been narrowed. + +The entry points in :mod:`hapi.run` used to declare their argument as +:class:`~hapi.catchment.Catchment` -- a class of 40-odd attributes, of which any one run touches +a dozen. That named the wrong thing: it over-stated the requirement, and pointed the dependency +at a concrete class, so the run layer could not be reasoned about without the class it runs. + +:class:`CatchmentLike` states the requirement instead, and states it *honestly*: a catchment is +a builder, so its inputs really are `X | None` until the matching `read_*` call has run. An +entry point accepts one of these and immediately narrows it with +:meth:`~hapi.runs.DistributedRun.from_model`, which is where the optionality is resolved and +every cross-input check happens. Past that seam the engines see +:class:`~hapi.runs.DistributedRun` or :class:`~hapi.runs.LumpedRun`, whose fields are not +optional at all. + +So there are two types on purpose, and the split is the point: this one describes what a caller +can hand over, the run types describe what an engine is allowed to receive. `Catchment` +satisfies this protocol structurally, so neither :mod:`hapi.run` nor :mod:`hapi.wrapper` imports +it at runtime. """ from __future__ import annotations @@ -21,75 +25,50 @@ import numpy as np -from hapi.conceptual import ConceptualModelSetup, ParameterSet -from hapi.inputs import FlowNetwork, MeteoInputs +from hapi.conceptual import ConceptualModelSetup, ParameterBounds, ParameterSet +from hapi.inputs import FlowNetwork, MeteoInputs, RiverGeometry from hapi.period import SimulationPeriod from hapi.results import SimulationResults -class ConceptualModelInputs(Protocol): - """What the per-cell conceptual model needs, whatever routes its output. +class CatchmentLike(Protocol): + """A catchment under assembly: the period is settled, the inputs may not be. - The part of the contract the lumped and distributed paths share. Split out so that - :class:`LumpedModel` does not advertise a need for a flow network it never touches. + Every field but `period` is optional, because that is the truth about a builder -- and it + is why an entry point narrows before it runs anything, rather than dereferencing these. Attributes: - parameters: The parameter array together with the `(snow, maxbas)` pair that fixes - its width -- one object, so the width rule is checked on every route to a set. - model_setup: The conceptual model instance, the catchment area and the state the - run starts from. - period: The span the run covers, and the calendar, `dt` and `conversion_factor` - it implies. One object rather than six loose fields, so a derived value can - never describe a different span from the one the model is set to. - results: Where the run writes its output. None before the first run. + period: The span the model covers. Settled at construction, so never `None`. + meteo: The three driver cubes, once assigned. + flow_network: The routing network and grid, once assigned. + parameters: The parameter set, once `read_parameters` has run. + model_setup: The conceptual model, once `read_lumped_model` has run. + data: The lumped driver record, once `read_lumped_inputs` has run. + river_geometry: The hydraulic rasters, once `read_river_geometry` has run. + flow_path_length_arr: The flow-path-length raster, once read. + bounds: The calibration search space, once `read_parameters_bound` has run. + routing_method: Which routing the parameter set was calibrated for. + results: Where a run's output lands. `None` before the first run. """ period: SimulationPeriod - parameters: ParameterSet - model_setup: ConceptualModelSetup + meteo: MeteoInputs | None + flow_network: FlowNetwork | None + parameters: ParameterSet | None + model_setup: ConceptualModelSetup | None + data: np.ndarray | None + river_geometry: RiverGeometry | None + flow_path_length_arr: np.ndarray | None + bounds: ParameterBounds | None + routing_method: str results: SimulationResults | None -class DistributedModel(ConceptualModelInputs, Protocol): - """What a distributed run requires on top of the conceptual model's own inputs. +class SupportsQsim(CatchmentLike, Protocol): + """A catchment that also has somewhere for a lumped hydrograph to land. Attributes: - meteo: The three driver cubes and the calendar they cover. - flow_network: The routing network and the grid it defines. + Qsim: Where `Run.run_lumped` puts the routed series, as a frame indexed by the period. """ - meteo: MeteoInputs - flow_network: FlowNetwork - - -class LumpedModelInputs(ConceptualModelInputs, Protocol): - """What a lumped run requires: one column per variable rather than a grid. - - Attributes: - data: `(time, 4)` array of precipitation, ET, temperature and the long-term average. - Qsim: Where the routed hydrograph lands. Whether MAXBAS routing applies is read off - `parameters.maxbas`, since that is what fixes the vector's width too. - """ - - data: np.ndarray Qsim: Any - - -class FloodModel(DistributedModel, Protocol): - """A distributed model that also carries the river geometry the flood model reads. - - Attributes: - routing_method: `"Kinematic"` when the kinematic-wave model routes the river cells, - which is what `run_flood` reads to decide whether the Muskingum pass skips them. - bankfull_depth: `(rows, cols)` bankfull depth. A positive value marks a river cell, - which the flood model can leave for a 1D hydraulic model to route. - river_width: `(rows, cols)` channel width. - river_roughness: `(rows, cols)` channel roughness. - flood_plain_roughness: `(rows, cols)` floodplain roughness. - """ - - routing_method: str - bankfull_depth: np.ndarray - river_width: np.ndarray - river_roughness: np.ndarray - flood_plain_roughness: np.ndarray diff --git a/src/hapi/rrm/distrrm.py b/src/hapi/rrm/distrrm.py index 140d8176..416d5587 100644 --- a/src/hapi/rrm/distrrm.py +++ b/src/hapi/rrm/distrrm.py @@ -15,6 +15,7 @@ from hapi.results import RoutingKind, SimulationResults from hapi.routing import Routing as routing +from hapi.runs import DistributedRun class DistributedRRM: @@ -24,8 +25,10 @@ class DistributedRRM: and routes the resulting discharge between cells following the river network. - The class is stateless; all methods are static and operate on a - `Model` object that carries the required arrays and parameters. + The class is stateless. Every method takes a :class:`~hapi.runs.DistributedRun` -- + validated, non-optional inputs -- and either returns the results it built or mutates the + :class:`~hapi.results.SimulationResults` it is handed. Nothing here reads or writes a + catchment, so nothing here has to ask whether its inputs were checked. """ def __init__(self): @@ -33,53 +36,21 @@ def __init__(self): pass @staticmethod - def run_lumped_model(Model) -> SimulationResults: + def run_lumped_model(run: DistributedRun) -> SimulationResults: """Run lumped rainfall-runoff model for every grid cell. - Executes the lumped conceptual model (e.g., HBV) independently - for each non-NaN cell in the catchment grid and converts the - resulting discharge from mm/time-step to m3/s. - - Builds `Model.results` and returns it, so a caller that needs the arrays does - not have to read them back off the model and re-narrow them from `| None`. + Args: + run: The validated inputs. Reads the flow network, the drivers, the parameter set + and the conceptual model setup. Returns: - SimulationResults: The results object, carrying `state_variables`, `quz` and - `qlz`, with `routing` still `RoutingKind.UNROUTED`. - - Args: - Model (Catchment): A catchment model object carrying the following - attributes: - - - `rows` (int): Number of grid rows. - - `cols` (int): Number of grid columns. - - `TS` (int): Number of time steps. - - `flow_acc_arr` (numpy.ndarray): 2-D flow accumulation - array; NaN marks cells outside the domain. - - `LumpedModel`: Lumped model instance with a - `simulate` method. - - `Prec` (numpy.ndarray): 3-D precipitation array - `(rows, cols, TS)`. - - `Temp` (numpy.ndarray): 3-D temperature array. - - `ET` (numpy.ndarray): 3-D evapotranspiration array. - - `ll_temp` (numpy.ndarray): 3-D long-term average - temperature array. - - `Parameters` (numpy.ndarray): 3-D parameter array - `(rows, cols, n_params)`. - - `InitialCond` (list): Initial state variable values - `[sp, sm, uz, lz, wc]`. - - `q_init` (float): Initial discharge in m3/s. - - `Snow` (int): Snow module flag (0 or 1). - - `CatArea` (float): Catchment area in km2. - - `px_tot_area` (float): Total pixel area in km2. - - `px_area` (float): Single pixel area in km2. - - `conversion_factor` (float): Unit conversion - factor (`tfac * 3.6`). + SimulationResults: A fresh results object carrying `state_variables`, `quz` and + `qlz`, with `routing` still `RoutingKind.UNROUTED` -- a routing step sets it. """ grid = ( - Model.flow_network.rows, - Model.flow_network.cols, - Model.meteo.simulation_steps, + run.flow_network.rows, + run.flow_network.cols, + run.meteo.simulation_steps, ) # A fresh results object per run, rather than nine attributes overwritten one at a # time: a half-finished run is then distinguishable from a finished one, and the @@ -90,74 +61,49 @@ def run_lumped_model(Model) -> SimulationResults: qlz=np.zeros(grid, dtype=np.float32), state_variables=np.zeros((*grid, 5), dtype=np.float32), ) - Model.results = results - for x in range(Model.flow_network.rows): - for y in range(Model.flow_network.cols): + for x in range(run.flow_network.rows): + for y in range(run.flow_network.cols): # only for cells in the domain - if not np.isnan(Model.flow_network.flow_acc_arr[x, y]): + if not np.isnan(run.flow_network.flow_acc_arr[x, y]): ( results.quz[x, y, :], results.qlz[x, y, :], results.state_variables[x, y, :, :], - ) = Model.model_setup.model.simulate( - prec=Model.meteo.precipitation[x, y, :], - temp=Model.meteo.temperature[x, y, :], - et=Model.meteo.evapotranspiration[x, y, :], - ll_temp=Model.meteo.ll_temp[x, y, :], - par=Model.parameters.values[x, y, :], - init_st=Model.model_setup.initial_cond, - q_init=Model.model_setup.q_init, - snow=Model.parameters.snow, + ) = run.model_setup.model.simulate( + prec=run.meteo.precipitation[x, y, :], + temp=run.meteo.temperature[x, y, :], + et=run.meteo.evapotranspiration[x, y, :], + ll_temp=run.meteo.ll_temp[x, y, :], + par=run.parameter_cube[x, y, :], + init_st=run.model_setup.initial_cond, + q_init=run.model_setup.q_init, + snow=run.parameters.snow, ) - area_coef = Model.model_setup.area / Model.flow_network.px_tot_area - factor = Model.flow_network.px_area * area_coef / Model.period.conversion_factor + area_coef = run.model_setup.area / run.flow_network.px_tot_area + factor = run.flow_network.px_area * area_coef / run.period.conversion_factor # convert quz and qlz from mm/time step to m3/sec # Timef*3.6 results.quz = results.quz * factor results.qlz = results.qlz * factor return results @staticmethod - def route_muskingum(Model, skip_hydraulic_cells: bool = False): + def route_muskingum(run: DistributedRun, results: SimulationResults) -> None: """Route discharge between cells following the flow direction. - Accumulates and routes upper-zone discharge (`quz`) using - Muskingum routing from upstream to downstream cells according - to the flow direction raster. Lower-zone discharge (`qlz`) - is translated (accumulated without attenuation) so that total - discharge can be computed at any internal point. - - After execution the following attributes are set on *Model*: - `quz_routed`, `qlz_translated`, and `q_total`. + Accumulates and routes upper-zone discharge from upstream to downstream cells along + the flow-direction network, and translates the lower zone (accumulated without + attenuation) so total discharge can be read at any internal point. Fills + `quz_routed`, `qlz_translated` and `q_total` on `results`, and records + `RoutingKind.MUSKINGUM`. Args: - Model (Catchment): A catchment model object carrying the following - attributes: - - - `rows` (int): Number of grid rows. - - `cols` (int): Number of grid columns. - - `TS` (int): Number of time steps. - - `flow_acc_arr` (numpy.ndarray): 2-D flow accumulation - array; NaN marks cells outside the domain. - - `quz` (numpy.ndarray): 3-D upper-zone discharge - array `(rows, cols, TS)` in m3/s. - - `qlz` (numpy.ndarray): 3-D lower-zone discharge - array `(rows, cols, TS)` in m3/s. - - `acc_val` (list): Sorted unique flow accumulation - values. - - `FDT` (dict): Flow direction table mapping - `"row,col"` keys to lists of upstream cell - index pairs. - - `Parameters` (numpy.ndarray): 3-D parameter array - where indices 10 and 11 are Muskingum K and X. - - `dt` (float): Time-step factor (`tfac`). - - `bankfull_depth` (numpy.ndarray): 2-D bankfull depth array. Read only - when `skip_hydraulic_cells` is True. - - skip_hydraulic_cells: Skip cells with a positive `bankfull_depth`, because a - 1D hydraulic model routes them instead. The flood model's path; a plain - distributed run leaves it False and routes every cell. + run: The validated inputs. `skip_hydraulic_cells` leaves cells with a positive + `river_geometry.bankfull_depth` unrouted, because a 1D hydraulic model routes + them instead; the run type has already checked the geometry is present. + results: The results to route, as returned by :meth:`run_lumped_model`. Mutated + in place. """ # # routing lake discharge with DS cell k & x and adding to cell Q # q_lake=Routing.muskingum_v(q_lake,q_lake[0],sp_pars[lakecell[0],lakecell[1],10],sp_pars[lakecell[0],lakecell[1],11],p2[0]) @@ -166,8 +112,14 @@ def route_muskingum(Model, skip_hydraulic_cells: bool = False): # #new # quz[lakecell[0],lakecell[1],:]=quz[lakecell[0],lakecell[1],:]+q_lake - results = Model.results # cells at the divider + # `DistributedRun` has already refused a skip with no geometry, so this is the value + # that guard proved is there -- bound once, outside the loop it is read in. + river_depth = ( + run.river_geometry.bankfull_depth + if run.skip_hydraulic_cells and run.river_geometry is not None + else None + ) results.quz_routed = np.zeros_like(results.quz) # lower zone discharge is going to be just translated without any attenuation @@ -176,29 +128,29 @@ def route_muskingum(Model, skip_hydraulic_cells: bool = False): results.qlz_translated = np.zeros_like(results.quz) # for all cells with 0 flow acc put the quz - for x in range(Model.flow_network.rows): # no of rows - for y in range(Model.flow_network.cols): # no of columns + for x in range(run.flow_network.rows): # no of rows + for y in range(run.flow_network.cols): # no of columns if ( - not np.isnan(Model.flow_network.flow_acc_arr[x, y]) - and Model.flow_network.flow_acc_arr[x, y] == 0 + not np.isnan(run.flow_network.flow_acc_arr[x, y]) + and run.flow_network.flow_acc_arr[x, y] == 0 ): results.quz_routed[x, y, :] = results.quz[x, y, :] results.qlz_translated[x, y, :] = results.qlz[x, y, :] # remaining cells # Read once: this is the routing inner loop, and `acc_val` scans the whole grid. - acc_val = Model.flow_network.acc_val + acc_val = run.flow_network.acc_val for j in range(1, len(acc_val)): # TODO parallelize # all cells with the same acc_val can run at the same time - for x in range(Model.flow_network.rows): # no of rows - for y in range(Model.flow_network.cols): # no of columns + for x in range(run.flow_network.rows): # no of rows + for y in range(run.flow_network.cols): # no of columns # check from total flow accumulation if ( - not np.isnan(Model.flow_network.flow_acc_arr[x, y]) - and Model.flow_network.flow_acc_arr[x, y] == acc_val[j] + not np.isnan(run.flow_network.flow_acc_arr[x, y]) + and run.flow_network.flow_acc_arr[x, y] == acc_val[j] ): - if skip_hydraulic_cells and Model.bankfull_depth[x, y] > 0: + if river_depth is not None and river_depth[x, y] > 0: # A river cell a 1D hydraulic model will route instead. The # caller says so explicitly; this used to be inferred from # `routing_method != "Muskingum"`, which meant any catchment @@ -208,28 +160,24 @@ def route_muskingum(Model, skip_hydraulic_cells: bool = False): continue else: # for UZ - q_uzi = np.zeros(Model.meteo.simulation_steps) + q_uzi = np.zeros(run.meteo.simulation_steps) # for lz - qlzi = np.zeros(Model.meteo.simulation_steps) + qlzi = np.zeros(run.meteo.simulation_steps) # iterate to route uz and translate lz for i in range( - len(Model.flow_network.FDT[str(x) + "," + str(y)]) - ): # Model.acc_val[j] + len(run.routing_table[str(x) + "," + str(y)]) + ): # bring the indexes of the us cell - x_ind = Model.flow_network.FDT[str(x) + "," + str(y)][ - i - ][0] - y_ind = Model.flow_network.FDT[str(x) + "," + str(y)][ - i - ][1] + x_ind = run.routing_table[str(x) + "," + str(y)][i][0] + y_ind = run.routing_table[str(x) + "," + str(y)][i][1] # sum the Q of the US cells (already routed for its cell) # route first with there own k & xthen sum q_uzi = q_uzi + routing.muskingum_v( results.quz_routed[x_ind, y_ind, :], results.quz_routed[x_ind, y_ind, 0], - Model.parameters.values[x_ind, y_ind, 10], - Model.parameters.values[x_ind, y_ind, 11], - Model.period.dt, + run.parameter_cube[x_ind, y_ind, 10], + run.parameter_cube[x_ind, y_ind, 11], + run.period.dt, ) qlzi = qlzi + results.qlz_translated[x_ind, y_ind, :] @@ -245,85 +193,68 @@ def route_muskingum(Model, skip_hydraulic_cells: bool = False): results.routing = RoutingKind.MUSKINGUM @staticmethod - def route_maxbas(Model): + def route_maxbas(run: DistributedRun, results: SimulationResults) -> None: """Route discharge to the outlet using a triangular function. - Applies triangular (MAXBAS) routing to the upper-zone - discharge of each cell independently. The MAXBAS parameter - is read from the last column of the spatially distributed - parameter array. - - The `Model.results.quz` array is modified in place. + Applies triangular (MAXBAS) routing to each cell's upper-zone discharge independently, + reading the MAXBAS parameter from the last column of the parameter array. `results.quz` + is modified in place. Args: - Model (Catchment): A catchment model object carrying the following - attributes: - - - `rows` (int): Number of grid rows. - - `cols` (int): Number of grid columns. - - `flow_acc_arr` (numpy.ndarray): 2-D flow accumulation - array; NaN marks cells outside the domain. - - `Parameters` (numpy.ndarray): 3-D parameter array - where the last index holds the MAXBAS value. - - `quz` (numpy.ndarray): 3-D upper-zone discharge - array `(rows, cols, TS)` in m3/s. + run: The validated inputs. + results: The results to route. Mutated in place. """ - Maxbas = Model.parameters.values[:, :, -1] - quz = Model.results.quz + Maxbas = run.parameter_cube[:, :, -1] + quz = results.quz - for x in range(Model.flow_network.rows): - for y in range(Model.flow_network.cols): - if not np.isnan(Model.flow_network.flow_acc_arr[x, y]): + for x in range(run.flow_network.rows): + for y in range(run.flow_network.cols): + if not np.isnan(run.flow_network.flow_acc_arr[x, y]): quz[x, y, :] = routing.triangular_routing_1( quz[x, y, :], Maxbas[x, y] ) @staticmethod - def route_maxbas_by_path_length(Model): + def route_maxbas_by_path_length( + run: DistributedRun, results: SimulationResults + ) -> None: """Route discharge using a triangular function scaled by flow path length. - Similar to `route_maxbas`, but the MAXBAS parameter for each - cell is rescaled proportionally to its flow path length so that - cells farther from the outlet receive more attenuation. - - The `Model.results.quz` array is modified in place. + Like :meth:`route_maxbas`, but each cell's MAXBAS is rescaled by its flow path length, + so cells farther from the outlet are attenuated more. `results.quz` is modified in + place. Args: - Model (Catchment): A catchment model object carrying the following - attributes: - - - `rows` (int): Number of grid rows. - - `cols` (int): Number of grid columns. - - `flow_acc_arr` (numpy.ndarray): 2-D flow accumulation - array; NaN marks cells outside the domain. - - `flow_path_length_arr` (numpy.ndarray): 2-D flow path length - array. - - `no_data_value` (float): No-data value used in the - flow path length raster. - - `Parameters` (numpy.ndarray): 3-D parameter array - where the last index holds the maximum MAXBAS value. - - `quz` (numpy.ndarray): 3-D upper-zone discharge - array `(rows, cols, TS)` in m3/s. + run: The validated inputs, whose `flow_path_length` supplies the raster. + results: The results to route. Mutated in place. + + Raises: + ValueError: The run carries no flow-path-length raster. """ - MAXBAS = np.nanmax(Model.parameters.values[:, :, -1]) + if run.flow_path_length is None: + raise ValueError( + "this routing scales MAXBAS by flow path length, but the run carries no " + "flow-path-length raster; call read_flow_path_length first" + ) + MAXBAS = np.nanmax(run.parameter_cube[:, :, -1]) # `read_flow_path_length` already masks this raster's own no-data cells to NaN via # pyramids, so no sentinel comparison is needed here -- and the one that used to sit # here compared against the *accumulation* raster's sentinel, which is a different # raster and need not share a no-data value. - MaxFPL = np.nanmax(Model.flow_path_length_arr) - MinFPL = np.nanmin(Model.flow_path_length_arr) + MaxFPL = np.nanmax(run.flow_path_length) + MinFPL = np.nanmin(run.flow_path_length) # resize_fun = lambda x: np.round(((((x - min_dist)/(max_dist - min_dist))*(1*maxbas - 1)) + 1), 0) resize_fun = lambda g: ( (((g - MinFPL) / (MaxFPL - MinFPL)) * (1 * MAXBAS - 1)) + 1 ) - NormalizedFPL = resize_fun(Model.flow_path_length_arr) - quz = Model.results.quz + NormalizedFPL = resize_fun(run.flow_path_length) + quz = results.quz - for x in range(Model.flow_network.rows): - for y in range(Model.flow_network.cols): - if not np.isnan(Model.flow_path_length_arr[x, y]): + for x in range(run.flow_network.rows): + for y in range(run.flow_network.cols): + if not np.isnan(run.flow_path_length[x, y]): quz[x, y, :] = routing.triangular_routing_2( quz[x, y, :], NormalizedFPL[x, y] ) diff --git a/src/hapi/run.py b/src/hapi/run.py index 5c9bb808..f0c8434c 100644 --- a/src/hapi/run.py +++ b/src/hapi/run.py @@ -22,12 +22,10 @@ import pandas as pd from loguru import logger -from hapi.protocols import ( - DistributedModel, - FloodModel, - LumpedModelInputs, -) +from hapi.inputs import RiverGeometry +from hapi.protocols import CatchmentLike, SupportsQsim from hapi.results import SimulationResults +from hapi.runs import DistributedRun, LumpedRun # from hapi.hm.saintvenant import SaintVenant from hapi.wrapper import Wrapper @@ -35,38 +33,12 @@ if TYPE_CHECKING: from hapi.catchment import Lake as LakeType -ROWS_MISMATCH_ERROR = "the parameters must have as many rows as the catchment grid" -COLS_MISMATCH_ERROR = "the parameters must have as many columns as the catchment grid" -#: The flow-direction check tests both axes, so it says both. The entry points used to -#: carry three different wordings for this one check; the widest is the accurate one. -GRID_MISMATCH_ERROR = "all input data should have the same number of rows and columns" - -def _check_parameters_cover_grid(model: DistributedModel) -> None: - """Check the parameter array spans the catchment grid. - - The same two checks every distributed entry point makes before handing the model to the - wrapper: a parameter array smaller than the grid is indexed out of range inside the - per-cell loop, far from the call that supplied it. - - Args: - model: The model about to run, carrying `parameters` and `flow_network`. - - Raises: - ValueError: The parameter array has the wrong number of rows or columns. - """ - shape = np.asarray(model.parameters.values).shape - if shape[0] != model.flow_network.rows: - raise ValueError(ROWS_MISMATCH_ERROR) - if shape[1] != model.flow_network.cols: - raise ValueError(COLS_MISMATCH_ERROR) - - -def _check_lake_meteo(model: DistributedModel, lake: LakeType) -> None: +def _check_lake_meteo(run: DistributedRun, lake: LakeType) -> None: """Check the lake's record lines up with the distributed drivers. Args: - model: The model about to run, whose `meteo` sets the expected length. + run: The validated run, whose `meteo` sets the expected length. lake: The lake whose `MeteoData` is checked. Raises: @@ -80,7 +52,7 @@ def _check_lake_meteo(model: DistributedModel, lake: LakeType) -> None: "the lake has no meteorological data; call lake.read_meteo_data before " "running a lake-aware entry point" ) - if np.shape(meteo_data)[0] != model.meteo.time_steps: + if np.shape(meteo_data)[0] != run.meteo.time_steps: raise ValueError( "Lake meteorological data has to have the same length as the distributed " "raster data" @@ -91,7 +63,9 @@ def _check_lake_meteo(model: DistributedModel, lake: LakeType) -> None: ) -def _warn_about_the_unrouted_river_cells(model: FloodModel) -> None: +def _warn_about_the_unrouted_river_cells( + run: DistributedRun, geometry: RiverGeometry +) -> None: """Warn that the river cells will be left unrouted, and by how much. Skipping them is the handoff the flood model was designed around: a 1D hydraulic model @@ -106,14 +80,17 @@ def _warn_about_the_unrouted_river_cells(model: FloodModel) -> None: statement that something downstream takes the river cells, and is left quiet. Args: - model: The model about to run, whose `bankfull_depth` marks the river cells. + run: The validated run, supplying the catchment mask. + geometry: The river geometry, whose `bankfull_depth` marks the river cells. Passed in + rather than read off `run`, where it is legitimately optional -- the caller has + already established it is present. """ # Only cells inside the catchment are ever routed, so only those can be skipped. The # bankfull-depth raster carries values outside the domain too, and counting those made # the message claim more river cells than the catchment has. - inside = ~np.isnan(model.flow_network.flow_acc_arr) + inside = ~np.isnan(run.flow_network.flow_acc_arr) river_cells = int( - np.count_nonzero((np.nan_to_num(model.bankfull_depth) > 0) & inside) + np.count_nonzero((np.nan_to_num(geometry.bankfull_depth) > 0) & inside) ) domain = int(np.count_nonzero(inside)) warnings.warn( @@ -127,39 +104,6 @@ def _warn_about_the_unrouted_river_cells(model: FloodModel) -> None: ) -def _validate_distributed(model: DistributedModel, check_flow_direction: bool) -> None: - """Run the checks every distributed entry point makes before the wrapper. - - Args: - model: The model about to run. - check_flow_direction: Whether to compare the flow-direction raster against the grid. - The MAXBAS paths never read that raster, so they do not require it. - - Raises: - ValueError: The grid, the drivers and the parameters do not agree. - """ - if check_flow_direction: - flow_dir_arr = model.flow_network.flow_dir_arr - # `FlowNetwork` takes the direction raster as optional because MAXBAS never reads - # it. The paths that route cell to cell do, so say which raster is missing rather - # than failing on None inside the routing loop. - if flow_dir_arr is None: - raise ValueError( - "this run routes cell to cell and needs a flow-direction raster, but the " - "flow network was built without one; pass it to FlowNetwork.from_rasters" - ) - fd_rows, fd_cols = flow_dir_arr.shape - if fd_rows != model.flow_network.rows or fd_cols != model.flow_network.cols: - raise ValueError(GRID_MISMATCH_ERROR) - - # The three cubes already agree with each other (checked when MeteoInputs was - # built); this is the other half -- that they cover the model's grid. - model.meteo.validate_against( - model.flow_network.rows, model.flow_network.cols, model.period.date_index - ) - _check_parameters_cover_grid(model) - - class Run: """Run the catchment model. @@ -199,7 +143,7 @@ class Run: """ @staticmethod - def run_distributed(model: DistributedModel) -> SimulationResults: + def run_distributed(model: CatchmentLike) -> SimulationResults: """Run the distributed hydrological model. Validates that all input arrays (precipitation, evapotranspiration, @@ -228,16 +172,16 @@ def run_distributed(model: DistributedModel) -> SimulationResults: ValueError: If input data arrays have inconsistent row counts, column counts, or temporal lengths. """ - _validate_distributed(model, check_flow_direction=True) - # run the model - results = Wrapper.run_muskingum(model) + run = DistributedRun.from_model(model) + results = Wrapper.run_muskingum(run) + model.results = results logger.info("Model Run has finished") return results @staticmethod def run_flood( - model: FloodModel, skip_hydraulic_cells: bool | None = None + model: CatchmentLike, skip_hydraulic_cells: bool | None = None ) -> SimulationResults: """Run the flood model. @@ -264,40 +208,25 @@ def run_flood( arrays, or river geometry arrays have inconsistent dimensions. """ - _validate_distributed(model, check_flow_direction=True) - - named_geometry = { - "bankfull_depth": model.bankfull_depth, - "river_width": model.river_width, - "river_roughness": model.river_roughness, - "flood_plain_roughness": model.flood_plain_roughness, - } - # `read_river_geometry` sets all four together, so a missing one means it was never - # called. Naming them beats `np.shape(None)` raising from inside the comparison. - missing = [name for name, arr in named_geometry.items() if arr is None] - if missing: - raise ValueError( - f"the flood model needs the river geometry, but {', '.join(missing)} " - "is not set; call read_river_geometry first" - ) - # Rebuilt from the non-None values rather than `.values()` directly: the guard above - # has already ruled None out, but only a comprehension carries that into the type. - geometry = [arr for arr in named_geometry.values() if arr is not None] - if any(np.shape(arr)[0] != model.flow_network.rows for arr in geometry): - raise ValueError(GRID_MISMATCH_ERROR) - if any(np.shape(arr)[1] != model.flow_network.cols for arr in geometry): - raise ValueError("all input data should have the same number of columns") - derived = skip_hydraulic_cells is None skip = ( - model.routing_method == "Kinematic" if derived else bool(skip_hydraulic_cells) + model.routing_method == "Kinematic" + if derived + else bool(skip_hydraulic_cells) ) - if skip and derived: - _warn_about_the_unrouted_river_cells(model) + # Every check the old inline block made now happens in the run type: `RiverGeometry` + # settles that the five rasters share a grid, and `DistributedRun` that the grid is the + # catchment's and that a requested skip has geometry to identify the river cells with. + run = DistributedRun.from_model( + model, with_river_geometry=True, skip_hydraulic_cells=skip + ) + + if skip and derived and run.river_geometry is not None: + _warn_about_the_unrouted_river_cells(run, run.river_geometry) - # run the model - results = Wrapper.run_muskingum(model, skip_hydraulic_cells=skip) + results = Wrapper.run_muskingum(run) + model.results = results logger.info("RRM has finished") # SV = SaintVenant() # SV.KinematicRaster(model) @@ -306,7 +235,7 @@ def run_flood( @staticmethod def run_distributed_with_lake( - model: DistributedModel, lake: LakeType + model: CatchmentLike, lake: LakeType ) -> SimulationResults: """Run the distributed model with a lake component. @@ -330,16 +259,16 @@ def run_distributed_with_lake( dimensions or if the lake meteorological data length does not match the distributed raster data length. """ - _validate_distributed(model, check_flow_direction=True) - _check_lake_meteo(model, lake) - # run the model - results = Wrapper.run_muskingum_with_lake(model, lake) + run = DistributedRun.from_model(model) + _check_lake_meteo(run, lake) + results = Wrapper.run_muskingum_with_lake(run, lake) + model.results = results logger.info("Model Run has finished") return results @staticmethod - def run_maxbas(model: DistributedModel) -> SimulationResults: + def run_maxbas(model: CatchmentLike) -> SimulationResults: """Run the FW1 distributed hydrological model. Validates that all input arrays have consistent dimensions, @@ -368,17 +297,15 @@ def run_maxbas(model: DistributedModel) -> SimulationResults: ValueError: If input data arrays have inconsistent row counts, column counts, or temporal lengths. """ - _validate_distributed(model, check_flow_direction=False) - # run the model - results = Wrapper.run_maxbas(model) + run = DistributedRun.from_model(model, needs_flow_direction=False) + results = Wrapper.run_maxbas(run) + model.results = results logger.info("Model Run has finished") return results @staticmethod - def run_maxbas_with_lake( - model: DistributedModel, lake: LakeType - ) -> SimulationResults: + def run_maxbas_with_lake(model: CatchmentLike, lake: LakeType) -> SimulationResults: """Run the FW1 distributed model with a lake component. Validates that all input arrays have consistent dimensions and @@ -400,15 +327,16 @@ def run_maxbas_with_lake( dimensions or if the lake meteorological data length does not match the distributed raster data length. """ - _validate_distributed(model, check_flow_direction=False) - _check_lake_meteo(model, lake) + run = DistributedRun.from_model(model, needs_flow_direction=False) + _check_lake_meteo(run, lake) - # run the model - return Wrapper.run_maxbas_with_lake(model, lake) + results = Wrapper.run_maxbas_with_lake(run, lake) + model.results = results + return results @staticmethod def run_lumped( - model: LumpedModelInputs, + model: SupportsQsim, Route: int = 0, routing_fn: Callable[..., Any] | None = None, ) -> SimulationResults: @@ -441,11 +369,15 @@ def run_lumped( # resolution -- this branch used to be written out here for the fourth time. ind = model.period.date_index - Qsim = pd.DataFrame(index=ind) + run = LumpedRun.from_model(model) + results = Wrapper.run_lumped(run, Route, routing_fn) - results = Wrapper.run_lumped(model, Route, routing_fn) - Qsim["q"] = model.Qsim + # The engine puts the lumped total in `results.q_total`; indexing it by the period and + # putting the frame on the model is this layer's job, not the engine's. + Qsim = pd.DataFrame(index=ind) + Qsim["q"] = results.q_total model.Qsim = Qsim[:] + model.results = results logger.info("Lumped model run has finished successfully") return results diff --git a/src/hapi/runs.py b/src/hapi/runs.py new file mode 100644 index 00000000..3f98eaa5 --- /dev/null +++ b/src/hapi/runs.py @@ -0,0 +1,277 @@ +"""The validated inputs one run needs, as a type rather than a convention. + +:class:`~hapi.catchment.Catchment` is a *builder*: it is constructed empty and filled by +`read_*` calls that may run in any order, some of which a given run never needs. So every input +on it is declared `X | None`, and that is honest -- a catchment half-way through assembly really +does have no flow network. + +The run layer needs the opposite thing: a catchment that is *finished*. Conflating the two is +what made the engines dereference `X | None` on every line, why four modules were excused from +mypy, and -- worse -- why "has this been validated?" was a question you answered by remembering +which entry point you came through. `Calibration` went straight to :class:`~hapi.wrapper.Wrapper` +and so skipped every check `Run` performed, on the one path that rebuilds the parameter array +thousands of times. + +:class:`DistributedRun` and :class:`LumpedRun` are that finished thing. Their fields are not +optional, and :meth:`DistributedRun.from_model` is the only way to get one: constructing it *is* +the validation. Nothing reaches the engines without passing through it, so the question stops +being one of discipline. + +They hold inputs only. Results come back as a return value -- see +:class:`~hapi.results.SimulationResults` -- so a run cannot half-overwrite what it read. +""" + +from __future__ import annotations + +from dataclasses import dataclass +from typing import TYPE_CHECKING, Any + +import numpy as np + +from hapi.conceptual import ConceptualModelSetup, ParameterSet +from hapi.inputs import FlowNetwork, MeteoInputs, RiverGeometry +from hapi.period import SimulationPeriod + +if TYPE_CHECKING: + from hapi.protocols import CatchmentLike + +ROWS_MISMATCH_ERROR = "the parameters must have as many rows as the catchment grid" +COLS_MISMATCH_ERROR = "the parameters must have as many columns as the catchment grid" +GRID_MISMATCH_ERROR = "all input data should have the same number of rows and columns" + + +def _require(model: CatchmentLike, name: str, hint: str) -> Any: + """Fetch an input a run cannot do without, or name the reader that supplies it. + + Args: + model: The catchment being narrowed. + name: Attribute to fetch. + hint: How to supply it, quoted in the error. + + Returns: + Any: The attribute's value. + + Raises: + ValueError: The attribute is unset. + """ + value = getattr(model, name, None) + if value is None: + raise ValueError(f"this run needs {name}, which is not set on the model; {hint}") + return value + + +@dataclass(frozen=True) +class DistributedRun: + """Everything a distributed run needs, checked and non-optional. + + Attributes: + period: The span the run covers, and the calendar and factors it implies. + meteo: The three driver cubes. + flow_network: The routing network and the grid it defines. + parameters: The parameter array plus the `(snow, maxbas)` pair fixing its width. + model_setup: The conceptual model instance and the state it starts from. + river_geometry: The five hydraulic rasters, when the flood path supplied them. `None` + on an ordinary distributed run, which never reads them. Absent-or-complete, never + half-filled -- that is what :class:`~hapi.inputs.RiverGeometry` guarantees. + skip_hydraulic_cells: Leave river cells unrouted for a 1D hydraulic model. Needs + `river_geometry` to identify them, checked here rather than in the routing loop. + flow_path_length: Flow-path length raster, read only by + :meth:`~hapi.rrm.distrrm.DistributedRRM.route_maxbas_by_path_length`. + """ + + period: SimulationPeriod + meteo: MeteoInputs + flow_network: FlowNetwork + parameters: ParameterSet + model_setup: ConceptualModelSetup + river_geometry: RiverGeometry | None = None + skip_hydraulic_cells: bool = False + flow_path_length: np.ndarray | None = None + + def __post_init__(self): + """Check the inputs agree with each other and with the grid. + + Raises: + ValueError: The drivers or the parameters do not cover the grid, the river geometry + does not, or a cell skip was asked for with no geometry to identify the river + cells. + """ + rows, cols = self.flow_network.rows, self.flow_network.cols + + # The three cubes already agree with each other (settled when MeteoInputs was built); + # this is the other half -- that they cover the grid, and the period. + self.meteo.validate_against(rows, cols, self.period.date_index) + + shape = np.asarray(self.parameters.values).shape + if shape[0] != rows: + raise ValueError(ROWS_MISMATCH_ERROR) + if shape[1] != cols: + raise ValueError(COLS_MISMATCH_ERROR) + + if self.river_geometry is not None and not self.river_geometry.covers(rows, cols): + raise ValueError(GRID_MISMATCH_ERROR) + + if self.skip_hydraulic_cells and self.river_geometry is None: + raise ValueError( + "skipping the hydraulic cells needs the river geometry to identify them, " + "but none is set; call read_river_geometry first" + ) + + @property + def parameter_cube(self) -> np.ndarray: + """np.ndarray: The parameters as the `(rows, cols, n)` cube a distributed run indexes. + + `ParameterSet.values` is a flat sequence for a lumped run and a cube for a distributed + one, so it is typed as either. On this side of the narrowing it is always the cube -- + `__post_init__` has already read its first two axes -- and saying so here means the + engines index a real array instead of a union. + """ + return np.asarray(self.parameters.values) + + @property + def routing_table(self) -> dict: + """dict: The flow-direction table, mapping `"row,col"` to the cells draining into it. + + `FlowNetwork` carries it as optional because the MAXBAS paths never route cell to cell. + Reaching it through here keeps that honest while giving the Muskingum routing a plain + dict to index. + + Raises: + ValueError: The network was built without a direction table. + """ + table = self.flow_network.FDT + if table is None: + raise ValueError( + "cell-to-cell routing needs the flow-direction table, but the flow network " + "was built without one; pass a flow-direction raster to " + "FlowNetwork.from_rasters" + ) + return table + + @classmethod + def from_model( + cls, + model: CatchmentLike, + *, + needs_flow_direction: bool = True, + with_river_geometry: bool = False, + skip_hydraulic_cells: bool = False, + ) -> DistributedRun: + """Narrow a built catchment into a validated distributed run. + + The single seam every distributed execution path passes through, `Run` and + `Calibration` alike. Everything checkable is checked here, so a caller cannot arrive at + the engines with an unvalidated model and does not have to remember to ask. + + Args: + model: A catchment with its inputs read. + needs_flow_direction: Whether the flow-direction raster is required. Cell-to-cell + routing needs it; MAXBAS sends every cell straight to the outlet and never + reads it. + with_river_geometry: Carry the river geometry through, for the flood path. + skip_hydraulic_cells: Leave the river cells to a hydraulic model. + + Returns: + DistributedRun: The validated inputs. + + Raises: + ValueError: A required input is unset, or the inputs disagree. + """ + flow_network = _require( + model, + "flow_network", + "assign FlowNetwork.from_rasters(...) to model.flow_network", + ) + if needs_flow_direction: + if flow_network.flow_dir_arr is None: + raise ValueError( + "this run routes cell to cell and needs a flow-direction raster, but the " + "flow network was built without one; pass it to FlowNetwork.from_rasters" + ) + # `FlowNetwork.__post_init__` already checks this at construction, but + # `__setattr__` does not re-check on replacement (unlike `MeteoInputs`), so a + # raster swapped in afterwards can still disagree with the grid. + if flow_network.flow_dir_arr.shape != (flow_network.rows, flow_network.cols): + raise ValueError(GRID_MISMATCH_ERROR) + if flow_network.FDT is None: + raise ValueError( + "cell-to-cell routing needs the flow-direction table; the flow network " + "was built without one" + ) + + geometry = None + if with_river_geometry or skip_hydraulic_cells: + geometry = _require(model, "river_geometry", "call read_river_geometry first") + + return cls( + period=model.period, + meteo=_require( + model, "meteo", "assign MeteoInputs.from_rasters(...) to model.meteo" + ), + flow_network=flow_network, + parameters=_require(model, "parameters", "call read_parameters first"), + model_setup=_require(model, "model_setup", "call read_lumped_model first"), + river_geometry=geometry, + skip_hydraulic_cells=skip_hydraulic_cells, + flow_path_length=getattr(model, "flow_path_length_arr", None), + ) + + +@dataclass(frozen=True) +class LumpedRun: + """Everything a lumped run needs, checked and non-optional. + + A lumped catchment has no grid, so it carries one column per driver rather than three cubes, + and no flow network at all. + + Attributes: + period: The span the run covers. + data: `(time, 4)` array of precipitation, ET, temperature and the long-term average. + parameters: The parameter vector plus the `(snow, maxbas)` pair fixing its width. + model_setup: The conceptual model instance and the state it starts from. + """ + + period: SimulationPeriod + data: np.ndarray + parameters: ParameterSet + model_setup: ConceptualModelSetup + + def __post_init__(self): + """Check the driver record covers the period the model was built for. + + Raises: + ValueError: The record is not four columns wide, or does not span the period. + """ + if np.ndim(self.data) != 2 or np.shape(self.data)[1] != 4: + raise ValueError( + "the lumped drivers must be a (time, 4) array of precipitation, ET, " + f"temperature and the long-term average, got shape {np.shape(self.data)}" + ) + steps = np.shape(self.data)[0] + if steps != len(self.period): + raise ValueError( + f"the lumped drivers hold {steps} steps but the model spans " + f"{len(self.period)} ({self.period.start:%Y-%m-%d} to " + f"{self.period.end:%Y-%m-%d}); the run is positional, so a mismatch silently " + "pairs each step with the wrong date" + ) + + @classmethod + def from_model(cls, model: CatchmentLike) -> LumpedRun: + """Narrow a built catchment into a validated lumped run. + + Args: + model: A catchment with its inputs read. + + Returns: + LumpedRun: The validated inputs. + + Raises: + ValueError: A required input is unset, or the record does not span the period. + """ + return cls( + period=model.period, + data=_require(model, "data", "call read_lumped_inputs first"), + parameters=_require(model, "parameters", "call read_parameters first"), + model_setup=_require(model, "model_setup", "call read_lumped_model first"), + ) diff --git a/src/hapi/wrapper.py b/src/hapi/wrapper.py index 29164f85..695b318b 100644 --- a/src/hapi/wrapper.py +++ b/src/hapi/wrapper.py @@ -13,16 +13,45 @@ import numpy as np -from hapi.protocols import ConceptualModelInputs, DistributedModel, LumpedModelInputs from hapi.results import RoutingKind, SimulationResults from hapi.routing import Routing as routing from hapi.rrm.distrrm import DistributedRRM as distrrm from hapi.rrm.hbv_lake import HBVLake +from hapi.runs import DistributedRun, LumpedRun if TYPE_CHECKING: from hapi.catchment import Lake +def _lake_inputs(lake: Lake) -> tuple[np.ndarray, list, list]: + """Fetch the two lake inputs the wrapper indexes, or say which reader supplies them. + + `Lake` is a builder like `Catchment`, so both are `X | None` until read. The wrapper used + to index them straight, so a lake missing either failed on `None` several frames in. + + Args: + lake: The lake about to be simulated. + + Returns: + tuple[np.ndarray, list, list]: The meteorological record, the parameter vector and the + outflow cell. + + Raises: + ValueError: Any of the three is unset. + """ + if lake.MeteoData is None: + raise ValueError( + "the lake has no meteorological data; call lake.read_meteo_data first" + ) + if lake.Parameters is None: + raise ValueError("the lake has no parameters; call lake.read_parameters first") + if lake.OutflowCell is None: + raise ValueError( + "the lake has no outflow cell; pass outflow_cell to lake.read_lumped_model" + ) + return lake.MeteoData, lake.Parameters, lake.OutflowCell + + class Wrapper: """Connects rainfall-runoff model components with spatial routing. @@ -45,12 +74,7 @@ def __init__(self): pass @staticmethod - def run_muskingum( - Model: DistributedModel, - ll_temp=None, - q_0=None, - skip_hydraulic_cells: bool = False, - ) -> SimulationResults: + def run_muskingum(run: DistributedRun) -> SimulationResults: """Run the distributed rainfall-runoff model with spatial routing. Connects two modules: @@ -89,21 +113,20 @@ def run_muskingum( The flood model's path. Defaults to False. Returns: - SimulationResults: The run's output, also assigned to `Model.results`. + SimulationResults: The run's output. Nothing is written to the caller's model; + the entry point in :mod:`hapi.run` is what puts it on `model.results`. """ # run the rainfall runoff model separately - results = distrrm.run_lumped_model(Model) + results = distrrm.run_lumped_model(run) # run the GIS part to rout from cell to another. It records # `RoutingKind.MUSKINGUM` on the results, which is what makes the outlet-cell # shortcut in `extract_discharge` valid for them. - distrrm.route_muskingum(Model, skip_hydraulic_cells=skip_hydraulic_cells) + distrrm.route_muskingum(run, results) return results @staticmethod - def run_muskingum_with_lake( - Model: DistributedModel, Lake: Lake, ll_temp=None, q_0=None - ) -> SimulationResults: + def run_muskingum_with_lake(run: DistributedRun, Lake: Lake) -> SimulationResults: """Run the distributed RRM with lake simulation and routing. Connects three modules: the lake module, the distributed @@ -134,18 +157,19 @@ def run_muskingum_with_lake( q_0 (float, optional): Initial discharge in m3/s. Defaults to None. """ - plake = Lake.MeteoData[:, 0] - et = Lake.MeteoData[:, 1] - t = Lake.MeteoData[:, 2] - tm = Lake.MeteoData[:, 3] + meteo_data, lake_parameters, outflow_cell = _lake_inputs(Lake) + plake = meteo_data[:, 0] + et = meteo_data[:, 1] + t = meteo_data[:, 2] + tm = meteo_data[:, 3] # lake simulation Lake.Qlake, _ = HBVLake().simulate( plake, t, et, - Lake.Parameters, - [Model.period.conversion_factor, Lake.CatArea, Lake.LakeArea], + lake_parameters, + [run.period.conversion_factor, Lake.CatArea, Lake.LakeArea], Lake.StageDischargeCurve, 0, init_st=Lake.InitialCond, @@ -157,21 +181,24 @@ def run_muskingum_with_lake( Lake.QlakeR = routing.muskingum_v( Lake.Qlake, Lake.Qlake[0], - Lake.Parameters[11], - Lake.Parameters[12], - Model.period.conversion_factor, + lake_parameters[11], + lake_parameters[12], + run.period.conversion_factor, ) # subcatchment - results = distrrm.run_lumped_model(Model) + results = distrrm.run_lumped_model(run) + # `ParameterSet.values` is a flat sequence for a lumped run and a cube for a + # distributed one; this path is distributed, so index it as the cube it is. + parameters = np.asarray(run.parameters.values) # routing lake discharge with DS cell k & x and adding to cell Q qlake = routing.muskingum_v( Lake.QlakeR, Lake.QlakeR[0], - Model.parameters.values[Lake.OutflowCell[0], Lake.OutflowCell[1], 10], - Model.parameters.values[Lake.OutflowCell[0], Lake.OutflowCell[1], 11], - Model.period.conversion_factor, + parameters[outflow_cell[0], outflow_cell[1], 10], + parameters[outflow_cell[0], outflow_cell[1], 11], + run.period.conversion_factor, ) # No padding: `HBVLake.simulate` already prepends the initial-state slot, exactly as @@ -181,17 +208,17 @@ def run_muskingum_with_lake( # input and left this entry point unrunnable. # both lake & Quz are in m3/s quz = results.quz - quz[Lake.OutflowCell[0], Lake.OutflowCell[1], :] = ( - quz[Lake.OutflowCell[0], Lake.OutflowCell[1], :] + qlake + quz[outflow_cell[0], outflow_cell[1], :] = ( + quz[outflow_cell[0], outflow_cell[1], :] + qlake ) # run the GIS part to rout from cell to another. It records # `RoutingKind.MUSKINGUM` on the results. - distrrm.route_muskingum(Model) + distrrm.route_muskingum(run, results) return results @staticmethod - def _set_maxbas_output_fields(Model: ConceptualModelInputs) -> None: + def _set_maxbas_output_fields(results: SimulationResults) -> None: """Fill the distributed output fields after a triangular (MAXBAS) run. `save_results` and `plot_distributed_results` read `q_total`, @@ -217,7 +244,6 @@ def _set_maxbas_output_fields(Model: ConceptualModelInputs) -> None: Model: Catchment whose `quz` / `qlz` have been routed by :meth:`DistRRM.route_maxbas`. """ - results = Model.results results.quz_routed = results.quz results.qlz_translated = results.qlz results.q_total = results.qlz + results.quz @@ -226,9 +252,7 @@ def _set_maxbas_output_fields(Model: ConceptualModelInputs) -> None: results.routing = RoutingKind.MAXBAS @staticmethod - def run_maxbas( - Model: DistributedModel, ll_temp=None, q_0=None - ) -> SimulationResults: + def run_maxbas(run: DistributedRun) -> SimulationResults: """Run the distributed RRM with triangular function-1 routing. Connects two modules: @@ -253,13 +277,13 @@ def run_maxbas( Defaults to None. """ # subcatchment - results = distrrm.run_lumped_model(Model) + results = distrrm.run_lumped_model(run) - distrrm.route_maxbas(Model) + distrrm.route_maxbas(run, results) - Wrapper._set_maxbas_output_fields(Model) + Wrapper._set_maxbas_output_fields(results) - steps = Model.meteo.simulation_steps + steps = run.meteo.simulation_steps qlz1 = np.array( [np.nansum(results.qlz[:, :, i]) for i in range(steps)] ) # average of all cells (not routed mm/timestep) @@ -271,9 +295,7 @@ def run_maxbas( return results @staticmethod - def run_maxbas_with_lake( - Model: DistributedModel, Lake: Lake, ll_temp=None, q_0=None - ) -> SimulationResults: + def run_maxbas_with_lake(run: DistributedRun, Lake: Lake) -> SimulationResults: """Run the distributed RRM with lake and triangular routing. Connects three modules: @@ -306,18 +328,19 @@ def run_maxbas_with_lake( q_0 (float, optional): Initial discharge in m3/s. Defaults to None. """ - plake = Lake.MeteoData[:, 0] - et = Lake.MeteoData[:, 1] - t = Lake.MeteoData[:, 2] - tm = Lake.MeteoData[:, 3] + meteo_data, lake_parameters, outflow_cell = _lake_inputs(Lake) + plake = meteo_data[:, 0] + et = meteo_data[:, 1] + t = meteo_data[:, 2] + tm = meteo_data[:, 3] # lake simulation Lake.Qlake, _ = HBVLake().simulate( plake, t, et, - Lake.Parameters, - [Model.period.conversion_factor, Lake.CatArea, Lake.LakeArea], + lake_parameters, + [run.period.conversion_factor, Lake.CatArea, Lake.LakeArea], Lake.StageDischargeCurve, 0, init_st=Lake.InitialCond, @@ -330,21 +353,21 @@ def run_maxbas_with_lake( Lake.QlakeR = routing.muskingum_v( Lake.Qlake, Lake.Qlake[0], - Lake.Parameters[11], - Lake.Parameters[12], - Model.period.conversion_factor, + lake_parameters[11], + lake_parameters[12], + run.period.conversion_factor, ) # subcatchment - results = distrrm.run_lumped_model(Model) + results = distrrm.run_lumped_model(run) - distrrm.route_maxbas(Model) + distrrm.route_maxbas(run, results) # Subcatchment fields only: the lake is a lumped inflow with no spatial # extent, so it enters `qout` below but never `q_total`. - Wrapper._set_maxbas_output_fields(Model) + Wrapper._set_maxbas_output_fields(results) - steps = Model.meteo.simulation_steps + steps = run.meteo.simulation_steps qlz1 = np.array( [np.nansum(results.qlz[:, :, i]) for i in range(steps)] ) # average of all cells (not routed mm/timestep) @@ -354,7 +377,7 @@ def run_maxbas_with_lake( qout = qlz1 + quz1 - # qout = (qlz1 + quz1) * Model.CatArea / (Model.period.conversion_factor* 3.6) + # qout = (qlz1 + quz1) * area / (run.period.conversion_factor * 3.6) # Both series run over `simulation_steps`, and the non-lake FW1 path returns # `qout[:-1]` -- dropping the trailing slot, not the leading initial-state one. The @@ -364,7 +387,7 @@ def run_maxbas_with_lake( @staticmethod def run_lumped( - Model: LumpedModelInputs, Routing: int = 0, RoutingFn: Callable | None = None + run: LumpedRun, Routing: int = 0, RoutingFn: Callable | None = None ) -> SimulationResults: """Run a lumped conceptual model with optional routing. @@ -412,32 +435,32 @@ def run_lumped( """ ### input data validation if Routing != 0: - if not callable(RoutingFn): + if RoutingFn is None or not callable(RoutingFn): raise TypeError( "routing function should be of type callable (function that takes " f"arguments), got {type(RoutingFn).__name__}" ) # data - p = Model.data[:, 0] - et = Model.data[:, 1] - t = Model.data[:, 2] - tm = Model.data[:, 3] + p = run.data[:, 0] + et = run.data[:, 1] + t = run.data[:, 2] + tm = run.data[:, 3] # from the conceptual model calculate the upper and lower response mm/time step - quz, qlz, state_variables = Model.model_setup.model.simulate( + quz, qlz, state_variables = run.model_setup.model.simulate( p, t, et, tm, - Model.parameters.values, - init_st=Model.model_setup.initial_cond, - q_init=Model.model_setup.q_init, - snow=Model.parameters.snow, + run.parameters.values, + init_st=run.model_setup.initial_cond, + q_init=run.model_setup.q_init, + snow=run.parameters.snow, ) # q mm , area sq km (1000**2)/1000/f/60/60 = 1/(3.6*f) # if daily tfac=24 if hourly tfac=1 if 15 min tfac=0.25 - factor = Model.model_setup.area / Model.period.conversion_factor + factor = run.model_setup.area / run.period.conversion_factor # A lumped run has no spatial routing at all, so the routed fields stay None and # the routing kind says why -- rather than a MAXBAS flag left over from elsewhere. results = SimulationResults( @@ -446,22 +469,26 @@ def run_lumped( qlz=qlz * factor, state_variables=state_variables, ) - Model.results = results - - Model.Qsim = results.quz + results.qlz - - if Routing != 0 and Model.parameters.maxbas: - Model.Qsim = RoutingFn( - np.array(Model.Qsim[:-1]), Model.parameters.values[-1] - ) + # The lumped total discharge is exactly what `q_total` means, so it goes there rather + # than onto the catchment as `Qsim`. `Run.run_lumped` is what indexes it by the period + # and puts the frame on the model -- so this engine writes nothing outside `results`. + q_total = results.quz + results.qlz + + if Routing != 0 and run.parameters.maxbas: + route = RoutingFn + assert route is not None # noqa: S101 - guarded above + q_total = route(np.array(q_total[:-1]), run.parameters.values[-1]) elif Routing != 0: - Model.Qsim = RoutingFn( - np.array(Model.Qsim[:-1]), - Model.Qsim[0], - Model.parameters.values[-2], - Model.parameters.values[-1], - Model.period.dt, + route = RoutingFn + assert route is not None # noqa: S101 - guarded above + q_total = route( + np.array(q_total[:-1]), + q_total[0], + run.parameters.values[-2], + run.parameters.values[-1], + run.period.dt, ) + results.q_total = q_total return results diff --git a/tests/rrm/catchment/test_fw1_output_fields.py b/tests/rrm/catchment/test_fw1_output_fields.py index d8ecc156..b9785e15 100644 --- a/tests/rrm/catchment/test_fw1_output_fields.py +++ b/tests/rrm/catchment/test_fw1_output_fields.py @@ -19,6 +19,7 @@ from hapi.rrm.distrrm import DistributedRRM from hapi.rrm.hbv_bergestrom92 import HBVBergestrom92 as HBVLumped from hapi.run import Run +from hapi.runs import DistributedRun DATE_REGEX = r"\d{4}.\d{2}.\d{2}" @@ -102,7 +103,9 @@ def coello_unrouted( coello.flow_network = FlowNetwork.from_rasters(coello_acc_path) coello.read_parameters(coello_dist_parameters_maxbas, False, maxbas=True) coello.read_lumped_model(HBVLumped, coello_cat_area, coello_initial_cond) - DistributedRRM.run_lumped_model(coello) + coello.results = DistributedRRM.run_lumped_model( + DistributedRun.from_model(coello, needs_flow_direction=False) + ) return coello diff --git a/tests/rrm/catchment/test_maxbas_routing_variants.py b/tests/rrm/catchment/test_maxbas_routing_variants.py index 572e09f6..9c641681 100644 --- a/tests/rrm/catchment/test_maxbas_routing_variants.py +++ b/tests/rrm/catchment/test_maxbas_routing_variants.py @@ -20,6 +20,7 @@ from hapi.rrm.distrrm import DistributedRRM from hapi.rrm.hbv_bergestrom92 import HBVBergestrom92 as HBVLumped from hapi.run import Run +from hapi.runs import DistributedRun, LumpedRun from hapi.wrapper import Wrapper @@ -66,7 +67,9 @@ def coello_before_routing( rows, cols = coello.flow_network.rows, coello.flow_network.cols distance = np.add.outer(np.arange(rows), np.arange(cols)).astype("float64") * 1000.0 coello.flow_path_length_arr = np.where(np.isnan(acc), np.nan, distance) - DistributedRRM.run_lumped_model(coello) + coello.results = DistributedRRM.run_lumped_model( + DistributedRun.from_model(coello, needs_flow_direction=False) + ) return coello @@ -106,7 +109,9 @@ def test_conserves_volume_while_redistributing_it_in_time( before = model.results.quz.copy() inside = ~np.isnan(model.flow_path_length_arr) - DistributedRRM.route_maxbas_by_path_length(model) + DistributedRRM.route_maxbas_by_path_length( + DistributedRun.from_model(model, needs_flow_direction=False), model.results + ) assert not np.array_equal(model.results.quz[inside], before[inside]), ( "the routing must alter the upper-zone discharge of the masked cells" @@ -143,7 +148,9 @@ def test_attenuates_the_peak_of_a_cell_far_from_the_outlet( ) before = model.results.quz.copy() - DistributedRRM.route_maxbas_by_path_length(model) + DistributedRRM.route_maxbas_by_path_length( + DistributedRun.from_model(model, needs_flow_direction=False), model.results + ) near_drop = before[nearest].max() - model.results.quz[nearest].max() far_drop = before[furthest].max() - model.results.quz[furthest].max() @@ -169,7 +176,9 @@ def test_leaves_cells_outside_the_mask_untouched( model.results.quz[outside] = sentinel before = model.results.quz.copy() - DistributedRRM.route_maxbas_by_path_length(model) + DistributedRRM.route_maxbas_by_path_length( + DistributedRun.from_model(model, needs_flow_direction=False), model.results + ) np.testing.assert_array_equal( model.results.quz[outside], @@ -224,9 +233,10 @@ def test_maxbas_routing_convolves_qsim_with_the_last_parameter( unrouted series, so compare against the same run left unrouted, convolved independently with the parameter the branch is supposed to use. """ - # Straight to the wrapper: `Run.run_lumped` wraps `Qsim` in a date-indexed frame and - # the unrouted series is one step longer than the index, so only the routed form - # survives that call. + # Straight to the wrapper: `Run.run_lumped` indexes the series by the period, and the + # unrouted one is a step longer than the index, so only the routed form survives that + # call. The engine leaves its total in `results.q_total`; `Qsim` is what the entry + # point puts on the model. unrouted = _lumped_model( coello_rrm_date, lumped_meteo_data_path, @@ -234,7 +244,7 @@ def test_maxbas_routing_convolves_qsim_with_the_last_parameter( coello_AreaCoeff, coello_InitialCond, ) - Wrapper.run_lumped(unrouted, Routing=0) + unrouted.results = Wrapper.run_lumped(LumpedRun.from_model(unrouted), Routing=0) routed = _lumped_model( coello_rrm_date, lumped_meteo_data_path, @@ -247,7 +257,7 @@ def test_maxbas_routing_convolves_qsim_with_the_last_parameter( maxbas = routed.parameters.values[-1] expected = Routing.triangular_routing_1( - np.array(np.asarray(unrouted.Qsim)[:-1]), maxbas + np.array(np.asarray(unrouted.results.q_total)[:-1]), maxbas ) # `run_lumped` wraps the routed series in a date-indexed frame; compare the values. actual = np.asarray(routed.Qsim, dtype=float).ravel() diff --git a/tests/rrm/catchment/test_run_narrowing.py b/tests/rrm/catchment/test_run_narrowing.py new file mode 100644 index 00000000..02c45549 --- /dev/null +++ b/tests/rrm/catchment/test_run_narrowing.py @@ -0,0 +1,249 @@ +"""Tests for the builder/finished split: narrowing a catchment into a validated run. + +`Catchment` is a builder -- its inputs are `X | None` until the matching `read_*` call has run, +and that is honest. The run layer needs the opposite: a catchment that is finished. Conflating +the two meant the engines dereferenced `X | None` on every line, and meant "has this been +validated?" was answered by remembering which entry point you came through -- which is how +`Calibration`, going straight to `Wrapper`, skipped every check `Run` performed. + +`DistributedRun.from_model` / `LumpedRun.from_model` are that seam. These tests pin the two +properties it buys: constructing the run *is* the validation, and there is no way to reach an +engine without doing it. +""" + +from __future__ import annotations + +import inspect + +import numpy as np +import pytest + +from hapi.catchment import Catchment +from hapi.inputs import FlowNetwork, MeteoInputs, RiverGeometry +from hapi.rrm.distrrm import DistributedRRM +from hapi.rrm.hbv_bergestrom92 import HBVBergestrom92 as HBVLumped +from hapi.runs import DistributedRun, LumpedRun +from hapi.wrapper import Wrapper + +DATE_REGEX = r"\d{4}.\d{2}.\d{2}" + + +@pytest.fixture +def built( + coello_start_date: str, + coello_end_date: str, + coello_prec_path: str, + coello_temp_path: str, + coello_evap_path: str, + coello_acc_path: str, + coello_fd_path: str, + coello_dist_parameters_muskingum: str, + coello_cat_area: int, + coello_initial_cond: list, +) -> Catchment: + """A distributed Coello catchment with every input read and no run behind it.""" + model = Catchment( + "coello", + coello_start_date, + coello_end_date, + spatial_resolution="Distributed", + temporal_resolution="Daily", + ) + model.meteo = MeteoInputs.from_rasters( + coello_prec_path, + coello_temp_path, + coello_evap_path, + start=coello_start_date, + end=coello_end_date, + regex_string=DATE_REGEX, + date=True, + file_name_data_fmt="%Y.%m.%d", + ) + model.flow_network = FlowNetwork.from_rasters(coello_acc_path, coello_fd_path) + model.read_parameters(coello_dist_parameters_muskingum, False) + model.read_lumped_model(HBVLumped, coello_cat_area, coello_initial_cond) + return model + + +class TestNarrowingIsTheValidation: + """`from_model` is where the optionality is resolved and the cross-checks happen.""" + + def test_a_finished_model_narrows(self, built: Catchment): + """Test that a fully built catchment produces a run with non-optional inputs. + + Test scenario: + The point of the split: past this seam nothing is `| None`, so the engines index + real arrays rather than unions and mypy can check them. + """ + run = DistributedRun.from_model(built) + + for field in ("period", "meteo", "flow_network", "parameters", "model_setup"): + assert getattr(run, field) is not None, f"{field} must be settled on the run" + assert run.meteo is built.meteo, "the run carries the model's own inputs, not copies" + + @pytest.mark.parametrize( + "missing, expected", + [ + ("meteo", "needs meteo"), + ("flow_network", "needs flow_network"), + ("parameters", "needs parameters"), + ("model_setup", "needs model_setup"), + ], + ) + def test_an_unread_input_is_named( + self, built: Catchment, missing: str, expected: str + ): + """Test that a missing input is reported by name, with the reader that supplies it. + + Test scenario: + A half-built catchment used to reach the engines and fail on `None` several frames + in, naming an attribute of an array rather than the reader nobody called. + + Args: + missing: The input to clear. + expected: Substring the error must carry. + """ + setattr(built, missing, None) + + with pytest.raises(ValueError, match=expected): + DistributedRun.from_model(built) + + def test_the_run_is_frozen(self, built: Catchment): + """Test that a narrowed run cannot be edited after it is checked. + + Test scenario: + The checks happen once, at construction. A mutable run would let a caller swap an + input in afterwards and reach the engines with something never validated -- the + exact hole this replaces. + """ + run = DistributedRun.from_model(built) + + with pytest.raises(Exception): # noqa: B017 - FrozenInstanceError + run.meteo = None + + def test_maxbas_does_not_require_a_flow_direction_raster(self, built: Catchment): + """Test that the optional input is only required by the paths that read it. + + Test scenario: + MAXBAS sends every cell straight to the outlet, so it never reads the direction + raster. Requiring it everywhere would refuse a legitimate run. + """ + built.flow_network = FlowNetwork( + built.flow_network.flow_acc_arr, + no_data_value=built.flow_network.no_data_value, + cell_size=built.flow_network.cell_size, + px_area=built.flow_network.px_area, + ) + + run = DistributedRun.from_model(built, needs_flow_direction=False) + + assert run.flow_network.flow_dir_arr is None, "the raster is genuinely absent" + with pytest.raises(ValueError, match="flow-direction"): + DistributedRun.from_model(built, needs_flow_direction=True) + + def test_a_skip_without_geometry_is_refused(self, built: Catchment): + """Test that asking to skip river cells with nothing to identify them raises. + + Test scenario: + The skip reads `bankfull_depth`. Without the geometry that used to be a + `TypeError` on `None` partway through the routing loop; the run type refuses it + before any cell is touched. + """ + with pytest.raises(ValueError, match="read_river_geometry"): + DistributedRun.from_model(built, skip_hydraulic_cells=True) + + def test_geometry_off_the_catchment_grid_is_refused(self, built: Catchment): + """Test that geometry on a different grid than the catchment raises. + + Test scenario: + `RiverGeometry` settles that the five rasters agree with *each other*; this is the + other half -- that they agree with the flow network. A cell index would otherwise + mean a different place in each. + """ + wrong = np.ones((built.flow_network.rows + 1, built.flow_network.cols)) + built.river_geometry = RiverGeometry(wrong, wrong, wrong, wrong, wrong) + + with pytest.raises(ValueError, match="same number of rows and columns"): + DistributedRun.from_model(built, with_river_geometry=True) + + +class TestTheEnginesCannotBeReachedUnvalidated: + """The seam is enforced by the signatures, not by remembering to call it.""" + + @pytest.mark.parametrize( + "func, expected", + [ + (DistributedRRM.run_lumped_model, DistributedRun), + (DistributedRRM.route_muskingum, DistributedRun), + (DistributedRRM.route_maxbas, DistributedRun), + (Wrapper.run_muskingum, DistributedRun), + (Wrapper.run_maxbas, DistributedRun), + (Wrapper.run_lumped, LumpedRun), + ], + ) + def test_every_engine_entry_takes_a_validated_run(self, func, expected): + """Test that each engine method's first parameter is a run type, not a catchment. + + Test scenario: + This is what makes the validation unskippable: `Calibration` used to call + `Wrapper` directly with a catchment and so bypassed every check. It cannot now -- + there is nothing to pass but a `DistributedRun` or a `LumpedRun`. + + Args: + func: The engine entry point. + expected: The run type its first parameter must be annotated with. + """ + first = next(iter(inspect.signature(func).parameters.values())) + + assert first.annotation in (expected, expected.__name__), ( + f"{func.__qualname__}'s first parameter must be {expected.__name__}, got " + f"{first.annotation!r}" + ) + + def test_the_engines_do_not_write_to_the_catchment(self, built: Catchment): + """Test that running the engine leaves the catchment untouched. + + Test scenario: + The engines used to assign results back onto the model they read. Returning them + instead is what lets the run type be frozen inputs, and means a run cannot + half-overwrite the object it was handed. + """ + run = DistributedRun.from_model(built) + + results = Wrapper.run_muskingum(run) + + assert built.results is None, ( + "the engine must not write to the catchment; the entry point in hapi.run is what " + "assigns model.results" + ) + assert results.q_total is not None, "the results come back as a return value" + + +class TestLumpedNarrowing: + """The lumped side gets the same treatment, against its own record shape.""" + + def test_a_driver_record_of_the_wrong_width_is_refused( + self, built: Catchment, coello_start_date: str, coello_end_date: str + ): + """Test that a record without the four driver columns raises. + + Test scenario: + `Wrapper.run_lumped` reads `data[:, 3]` -- the long-term average -- so a + three-column record fails inside the run. The shape is settled up front instead. + """ + built.data = np.ones((len(built.period), 3)) + + with pytest.raises(ValueError, match=r"\(time, 4\) array"): + LumpedRun.from_model(built) + + def test_a_record_that_does_not_span_the_period_is_refused(self, built: Catchment): + """Test that a record of the wrong length raises rather than misaligning silently. + + Test scenario: + The run is positional, so a record shorter or longer than the period pairs each + step with the wrong date and still produces numbers. + """ + built.data = np.ones((len(built.period) + 5, 4)) + + with pytest.raises(ValueError, match="the run is positional"): + LumpedRun.from_model(built) diff --git a/tests/rrm/catchment/test_run_results_coupling.py b/tests/rrm/catchment/test_run_results_coupling.py index e075bd12..5df8d8cc 100644 --- a/tests/rrm/catchment/test_run_results_coupling.py +++ b/tests/rrm/catchment/test_run_results_coupling.py @@ -372,16 +372,19 @@ def test_the_flood_model_names_the_river_geometry_it_lacks( """Test that the flood model reports which geometry rasters are unset. Test scenario: - `read_river_geometry` sets four arrays together, so a missing one means it was - never called. The check used to reach `np.shape(None)`; it now lists the names. + The five rasters are one `RiverGeometry` now, built and checked together, so the + flood path either has it or does not -- and says which reader supplies it. The + check used to reach `np.shape(None)` on whichever array was missing. """ model = _build("coello", coello_dist_parameters_muskingum, **coello_fixtures) with pytest.raises(ValueError, match="read_river_geometry") as exc_info: Run.run_flood(model) - assert "bankfull_depth" in str(exc_info.value), ( - f"the error should name the missing rasters, got: {exc_info.value}" + # One object to report, not four rasters: `RiverGeometry` is absent-or-complete, so + # there is no longer a "three of the four are set" state to enumerate. + assert "river_geometry" in str(exc_info.value), ( + f"the error should name the missing input, got: {exc_info.value}" ) diff --git a/tests/rrm/catchment/test_run_validation.py b/tests/rrm/catchment/test_run_validation.py index 242a069d..ac7f20aa 100644 --- a/tests/rrm/catchment/test_run_validation.py +++ b/tests/rrm/catchment/test_run_validation.py @@ -15,9 +15,10 @@ from hapi import run as run_module from hapi.catchment import Catchment -from hapi.inputs import FlowNetwork, MeteoInputs +from hapi.inputs import FlowNetwork, MeteoInputs, RiverGeometry from hapi.rrm.hbv_bergestrom92 import HBVBergestrom92 as HBVLumped from hapi.run import Run +from hapi.runs import DistributedRun class _LakeStub: @@ -98,10 +99,13 @@ def _load_flat_river_geometry(model: Catchment) -> None: model: Model whose `flow_network` supplies the grid shape. """ shape = (model.flow_network.rows, model.flow_network.cols) - model.bankfull_depth = np.full(shape, 2.0) - model.river_width = np.full(shape, 10.0) - model.river_roughness = np.full(shape, 0.03) - model.flood_plain_roughness = np.full(shape, 0.06) + model.river_geometry = RiverGeometry( + dem=np.full(shape, 100.0), + bankfull_depth=np.full(shape, 2.0), + river_width=np.full(shape, 10.0), + river_roughness=np.full(shape, 0.03), + flood_plain_roughness=np.full(shape, 0.06), + ) class TestRunFloodModel: @@ -124,8 +128,12 @@ def test_dispatches_once_every_input_lines_up( assert "run_muskingum" in spied_wrapper, ( "run_muskingum should have been dispatched" ) - assert spied_wrapper["run_muskingum"][0] is coello_loaded, ( - "the wrapper must receive the model itself" + run = spied_wrapper["run_muskingum"][0] + assert isinstance(run, DistributedRun), ( + f"the wrapper must receive a validated DistributedRun, got {type(run).__name__}" + ) + assert run.meteo is coello_loaded.meteo, ( + "the run must carry the model's own inputs, not copies" ) def test_rejects_river_geometry_off_the_catchment_grid( @@ -138,11 +146,13 @@ def test_rejects_river_geometry_off_the_catchment_grid( flow network. A width raster one row short must raise rather than index past the end of the grid inside the hydraulic routing. """ - _load_flat_river_geometry(coello_loaded) - coello_loaded.river_width = coello_loaded.river_width[:-1, :] + shape = (coello_loaded.flow_network.rows, coello_loaded.flow_network.cols) + flat = np.full(shape, 1.0) - with pytest.raises(ValueError, match="number of rows"): - Run.run_flood(coello_loaded) + # `RiverGeometry` refuses the set outright: the five must describe one grid, and that + # is settled where they are built rather than inside the flood entry point. + with pytest.raises(ValueError, match="must share one grid"): + RiverGeometry(flat, flat, flat[:-1, :], flat, flat) assert "run_muskingum" not in spied_wrapper, ( "the wrapper must not run on inconsistent geometry" @@ -194,8 +204,9 @@ def test_dispatches_once_the_lake_record_matches_the_simulation( assert "run_muskingum_with_lake" in spied_wrapper, ( "run_muskingum_with_lake should have been dispatched" ) - assert spied_wrapper["run_muskingum_with_lake"] == (coello_loaded, lake), ( - "the wrapper must receive the model and the lake" + run, seen_lake = spied_wrapper["run_muskingum_with_lake"] + assert isinstance(run, DistributedRun) and seen_lake is lake, ( + "the wrapper must receive a validated DistributedRun and the lake" ) def test_rejects_a_lake_record_of_the_wrong_length( @@ -271,8 +282,9 @@ def test_dispatches_once_the_lake_record_matches_the_simulation( assert "run_maxbas_with_lake" in spied_wrapper, ( "run_maxbas_with_lake should have been dispatched" ) - assert spied_wrapper["run_maxbas_with_lake"] == (coello_loaded, lake), ( - "the wrapper must receive the model and the lake" + run, seen_lake = spied_wrapper["run_maxbas_with_lake"] + assert isinstance(run, DistributedRun) and seen_lake is lake, ( + "the wrapper must receive a validated DistributedRun and the lake" ) def test_rejects_parameters_off_the_catchment_grid( @@ -326,8 +338,8 @@ def test_the_skip_is_derived_from_the_declared_routing( coello_loaded.routing_method = declared seen: dict = {} - def _spy(model, ll_temp=None, q_0=None, skip_hydraulic_cells=False): - seen["skip"] = skip_hydraulic_cells + def _spy(run): + seen["skip"] = run.skip_hydraulic_cells monkeypatch.setattr(run_module.Wrapper, "run_muskingum", staticmethod(_spy)) @@ -351,8 +363,8 @@ def test_an_explicit_argument_overrides_the_declaration( coello_loaded.routing_method = "Muskingum" seen: dict = {} - def _spy(model, ll_temp=None, q_0=None, skip_hydraulic_cells=False): - seen["skip"] = skip_hydraulic_cells + def _spy(run): + seen["skip"] = run.skip_hydraulic_cells monkeypatch.setattr(run_module.Wrapper, "run_muskingum", staticmethod(_spy)) diff --git a/tests/rrm/catchment/test_wrapper_with_lake.py b/tests/rrm/catchment/test_wrapper_with_lake.py index 43c678d9..e17997a6 100644 --- a/tests/rrm/catchment/test_wrapper_with_lake.py +++ b/tests/rrm/catchment/test_wrapper_with_lake.py @@ -21,6 +21,7 @@ from hapi.rrm.hbv_bergestrom92 import HBVBergestrom92 as HBVLumped from hapi.rrm.hbv_lake import HBVLake from hapi.run import Run +from hapi.runs import DistributedRun from hapi.wrapper import Wrapper JIBOA_ROOT = "tests/rrm/data/jiboa" @@ -296,8 +297,12 @@ def test_the_lake_raises_discharge_at_the_outflow_cell( model = coello_with_lake_inputs lake = _make_lake(model, coello_start_date, coello_end_date, seed=7) - Wrapper.run_muskingum(coello_no_lake) - Wrapper.run_muskingum_with_lake(model, lake) + coello_no_lake.results = Wrapper.run_muskingum( + DistributedRun.from_model(coello_no_lake) + ) + model.results = Wrapper.run_muskingum_with_lake( + DistributedRun.from_model(model), lake + ) row, col = OUTFLOW_CELL without = coello_no_lake.results.quz_routed[row, col, :] @@ -346,7 +351,9 @@ def test_the_routed_lake_series_covers_every_simulation_step( model = coello_with_lake_inputs lake = _make_lake(model, coello_start_date, coello_end_date, seed=11) - Wrapper.run_muskingum_with_lake(model, lake) + model.results = Wrapper.run_muskingum_with_lake( + DistributedRun.from_model(model), lake + ) steps = model.meteo.simulation_steps rows, cols = model.flow_network.rows, model.flow_network.cols @@ -381,13 +388,17 @@ def test_a_muskingum_lake_run_clears_a_flag_a_triangular_run_set( model = coello_with_lake_inputs_maxbas lake = _make_lake(model, coello_start_date, coello_end_date, seed=13) - Wrapper.run_maxbas_with_lake(model, lake) + model.results = Wrapper.run_maxbas_with_lake( + DistributedRun.from_model(model, needs_flow_direction=False), lake + ) assert not model.results.outlet_shortcut_valid, ( "the triangular path must mark the model before this test means anything" ) model.read_parameters(coello_dist_parameters_muskingum, False) - Wrapper.run_muskingum_with_lake(model, lake) + model.results = Wrapper.run_muskingum_with_lake( + DistributedRun.from_model(model), lake + ) assert model.results.outlet_shortcut_valid, ( "a Muskingum lake run makes the outlet-cell shortcut valid again" @@ -415,7 +426,9 @@ def test_fills_the_distributed_output_fields( lake = _make_lake(model, coello_start_date, coello_end_date, seed=17) assert model.results is None, "the fixture must arrive with no run behind it" - Wrapper.run_maxbas_with_lake(model, lake) + model.results = Wrapper.run_maxbas_with_lake( + DistributedRun.from_model(model, needs_flow_direction=False), lake + ) rows, cols = model.flow_network.rows, model.flow_network.cols steps = model.meteo.simulation_steps @@ -443,7 +456,9 @@ def test_the_outlet_series_carries_the_lake_and_drops_the_extra_slot( model = coello_with_lake_inputs_maxbas lake = _make_lake(model, coello_start_date, coello_end_date, seed=19) - Wrapper.run_maxbas_with_lake(model, lake) + model.results = Wrapper.run_maxbas_with_lake( + DistributedRun.from_model(model, needs_flow_direction=False), lake + ) expected_len = model.meteo.simulation_steps - 1 assert len(model.results.qout) == expected_len, ( @@ -481,7 +496,9 @@ def test_marks_the_model_as_maxbas_routed( lake = _make_lake(model, coello_start_date, coello_end_date, seed=23) assert model.results is None, "a fresh model must arrive with no results" - Wrapper.run_maxbas_with_lake(model, lake) + model.results = Wrapper.run_maxbas_with_lake( + DistributedRun.from_model(model, needs_flow_direction=False), lake + ) assert not model.results.outlet_shortcut_valid, ( "a triangular lake run must mark the model as MAXBAS-routed" From 11fe3005ca82bc3611e67c65b0a0ac5e8d1c8793 Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Sun, 6 Sep 2026 19:17:05 +0200 Subject: [PATCH 11/54] refactor(calibration)!: hold the catchment instead of being one `Calibration` subclassed `Catchment`, which meant it inherited a forty-attribute builder in order to use a dozen fields of it, and inherited `plot_hydrograph` -- which reads `Qsim.loc[...]` and so could never work against the bare array this class's own `extract_discharge` produces. That was a Liskov break with a concrete failure, and it was reachable. Composition does not fix it so much as make it impossible: nothing is inherited, so nothing can be inherited broken. `Calibration(model)` now takes the catchment it calibrates, which also means a model from `Catchment.from_yaml` can be handed straight over. The search space moves with it. `ParameterBounds` is read by nothing but a calibration, and it carries the `(snow, maxbas)` pair every trial vector is checked against, so `read_parameters_bound` and `bounds` belong beside the optimiser rather than on the model being optimised. `Catchment.__init__` is down to 19 attributes, from 42 when this started. Two bugs surfaced while doing it, both of the same kind and both now fixed. The narrowing added in the previous commit was inside the objective function's bare `except:`, so a validation failure was caught and scored `nan` -- the optimiser then searched on over a model that never ran. The narrowing now happens *outside* the try, because a wrong-width vector or a grid mismatch is a defect to surface rather than an infeasible candidate; and the three bare `except:` clauses are `except Exception` with a warning naming what was swallowed, so a long calibration can also be interrupted again. That immediately exposed the second: `calibrate_maxbas` was calling `Wrapper.run_maxbas(run)` with `run` never assigned, and its tests passed because the `NameError` was swallowed as an infeasible trial. Two `TestLumpedCalibration` tests were passing the same way -- they never called `read_lumped_model`, so every trial failed while the stubbed optimiser returned its canned result regardless. Both are fixed rather than papered over. `hapi.calibration` stays on the mypy suppression list, and the comment now says why rather than promising assert helpers: it reads the builder's genuinely optional attributes, and its `SpatialVarFun` argument is typed `Callable` though the code uses `.Function` / `.Par3d` / `.no_parameters` on it. A Protocol for that contract would close most of the remaining 44 errors. BREAKING CHANGE: `Calibration(name, start, end, ...)` becomes `Calibration(Catchment(name, start, end, ...))`, and `Calibration.from_yaml` is gone -- use `Calibration(Catchment.from_yaml(path))`. Everything inherited from `Catchment` is reached through `.model`: `calibration.model.read_parameters(...)`, `calibration.model.meteo`, `calibration.model.QGauges`. `read_parameters_bound` and `bounds` move from `Catchment` to `Calibration`. --- ...stributed-model-calibratation-muskingum.py | 21 +- ...libration-deap-multiobjective-NSE-NSEHF.py | 27 +- ...alibration-deap-multiobjective-NSE-RMSE.py | 27 +- .../coello-lumped-model-calibration-deap.py | 25 +- .../coello-lumped-model-calibration.py | 17 +- ...stributed-model-calibratation-muskingum.py | 23 +- pyproject.toml | 13 +- src/hapi/calibration.py | 275 ++++++++++-------- src/hapi/catchment.py | 51 +--- src/hapi/runs.py | 17 +- tests/calibration/distributed_mode_calib.py | 19 +- tests/calibration/lumped_calibration.py | 17 +- .../test_calibration_distributed.py | 77 +++-- tests/rrm/calibration/test_rrm_calibration.py | 90 ++++-- tests/rrm/catchment/test_config.py | 16 +- tests/rrm/catchment/test_rrm_catchment.py | 23 -- tests/rrm/catchment/test_run_narrowing.py | 8 +- 17 files changed, 409 insertions(+), 337 deletions(-) diff --git a/examples/hydrological-model/coello/calibration/coello-hrus-distributed-model-calibratation-muskingum.py b/examples/hydrological-model/coello/calibration/coello-hrus-distributed-model-calibratation-muskingum.py index e023a852..60627fc0 100644 --- a/examples/hydrological-model/coello/calibration/coello-hrus-distributed-model-calibratation-muskingum.py +++ b/examples/hydrological-model/coello/calibration/coello-hrus-distributed-model-calibratation-muskingum.py @@ -9,6 +9,7 @@ from statista.descriptors import rmse from hapi.calibration import Calibration +from hapi.catchment import Catchment from hapi.inputs import FlowNetwork, MeteoInputs from hapi.rrm.hbv_bergestrom92 import HBVBergestrom92 as HBV from hapi.rrm.parameters import Parameters as DP @@ -33,10 +34,12 @@ start_date = "2009-01-01" end_date = "2011-12-31" name = "Coello" -Coello = Calibration(name, start_date, end_date, spatial_resolution="Distributed") -Coello.meteo = MeteoInputs.from_rasters(PrecPath, TempPath, Evap_Path) -Coello.flow_network = FlowNetwork.from_rasters(FlowAccPath, FlowDPath) -Coello.read_lumped_model(HBV, AreaCoeff, InitialCond) +Coello = Calibration( + Catchment(name, start_date, end_date, spatial_resolution="Distributed") +) +Coello.model.meteo = MeteoInputs.from_rasters(PrecPath, TempPath, Evap_Path) +Coello.model.flow_network = FlowNetwork.from_rasters(FlowAccPath, FlowDPath) +Coello.model.read_lumped_model(HBV, AreaCoeff, InitialCond) # %% UB = np.loadtxt(path + "/Basic_inputs/UB_HRU.txt", usecols=0) LB = np.loadtxt(path + "/Basic_inputs/LB_HRU.txt", usecols=0) @@ -82,11 +85,11 @@ # this nomber is just an indication to prepare the UB & LB file don't input it to the model # SpatialVarArgs=[raster,no_parameters,no_lumped_par,lumped_par_pos] # %% Gauges -Coello.read_gauge_table(path + "Discharge/stations/gauges.csv", FlowAccPath) +Coello.model.read_gauge_table(path + "Discharge/stations/gauges.csv", FlowAccPath) GaugesPath = path + "Discharge/stations/" -Coello.read_discharge_gauges(GaugesPath, column="id", fmt="%Y-%m-%d") +Coello.model.read_discharge_gauges(GaugesPath, column="id", fmt="%Y-%m-%d") # %% Objective function -coordinates = Coello.GaugesTable[["id", "x", "y", "weight"]][:] +coordinates = Coello.model.GaugesTable[["id", "x", "y", "weight"]][:] # define the objective function and its arguments OF_args = [coordinates] @@ -136,5 +139,7 @@ def objective_function(q_obs, coordinates): # Qout, q_uz_routed, q_lz_trans, # %% run calibration cal_parameters = Coello.run_calibration(SpatialVarFun, OptimizationArgs, print_error=0) # %% convert parameters to rasters -SpatialVarFun.Function(Coello.parameters, kub=SpatialVarFun.Kub, klb=SpatialVarFun.Klb) +SpatialVarFun.Function( + Coello.model.parameters, kub=SpatialVarFun.Kub, klb=SpatialVarFun.Klb +) SpatialVarFun.save_parameters(SaveTo) diff --git a/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-NSEHF.py b/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-NSEHF.py index 4dc21069..7248b8c7 100644 --- a/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-NSEHF.py +++ b/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-NSEHF.py @@ -16,6 +16,7 @@ from deap import algorithms, base, creator, tools from hapi.calibration import Calibration +from hapi.catchment import Catchment from hapi.routing import Routing from hapi.rrm.hbv_bergestrom92 import HBVBergestrom92 as HBVLumped from hapi.run import Run @@ -30,8 +31,8 @@ end = "2011-12-31" name = "Coello" -Coello = Calibration(name, start, end) -Coello.read_lumped_inputs(MeteoDataPath) +Coello = Calibration(Catchment(name, start, end)) +Coello.model.read_lumped_inputs(MeteoDataPath) ### Basic_inputs @@ -42,7 +43,7 @@ InitialCond = [0, 10, 10, 10, 0] # no snow subroutine Snow = False -Coello.read_lumped_model(HBVLumped, AreaCoeff, InitialCond) +Coello.model.read_lumped_model(HBVLumped, AreaCoeff, InitialCond) # Calibration parameters @@ -64,7 +65,7 @@ RoutingFn = Routing.triangular_routing_1 # outlet discharge -Coello.read_discharge_gauges(Path + "Qout_c.csv", fmt="%Y-%m-%d") +Coello.model.read_discharge_gauges(Path + "Qout_c.csv", fmt="%Y-%m-%d") # %% Calibration creator.create("Fitness", base.Fitness, weights=(1.0, 1.0)) creator.create("IndividualContainer", list, fitness=creator.Fitness) @@ -93,12 +94,12 @@ def initializer(): def objfn(individual): - # Coello.read_parameters(Parameterpath, Snow) - Coello.parameters = individual + # Coello.model.read_parameters(Parameterpath, Snow) + Coello.model.parameters = individual Run.run_lumped(Coello, Route, RoutingFn) - # [Coello.QGauges.columns[-1]] - NSE = metrics.nse_hf(Coello.QGauges, Coello.Qsim, *Coello.OFArgs) - NSEHF = metrics.nse_hf(Coello.QGauges, Coello.Qsim, *Coello.OFArgs) + # [Coello.model.QGauges.columns[-1]] + NSE = metrics.nse_hf(Coello.model.QGauges, Coello.Qsim, *Coello.OFArgs) + NSEHF = metrics.nse_hf(Coello.model.QGauges, Coello.Qsim, *Coello.OFArgs) return NSE, NSEHF @@ -147,7 +148,7 @@ def distance(individual): print("Best individual is %s, %s" % (best_ind, best_ind.fitness.values)) # %% Run the Model -Coello.parameters = best_ind +Coello.model.parameters = best_ind # [0.7686518278956287, 144.35510831203874, 1.9922719933560913, 0.1439126168555068, 0.9474744708723734, # 0.749219030317463, 0.8074091462437563, 0.07289588281400794, 68.83482640397304, 5.123384184968337, # 1.9922719933560913] @@ -157,7 +158,7 @@ def distance(individual): scores = dict() -Qobs = Coello.QGauges[Coello.QGauges.columns[0]] +Qobs = Coello.model.QGauges[Coello.model.QGauges.columns[0]] scores["RMSE"] = metrics.rmse(Qobs, Coello.Qsim["q"]) scores["NSE"] = metrics.nse(Qobs, Coello.Qsim["q"]) @@ -175,7 +176,7 @@ def distance(individual): gaugei = 0 plotstart = "2009-01-01" plotend = "2011-12-31" -Coello.plot_hydrograph(plotstart, plotend, gaugei, title="Lumped Model") +Coello.model.plot_hydrograph(plotstart, plotend, gaugei, title="Lumped Model") # %% Save the Parameters @@ -200,4 +201,4 @@ def distance(individual): + str(dt.datetime.now())[0:10] + ".txt" ) -Coello.save_results(result=5, start=StartDate, end=EndDate, path=Path) +Coello.model.save_results(result=5, start=StartDate, end=EndDate, path=Path) diff --git a/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-RMSE.py b/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-RMSE.py index 3843bb50..03d18d2e 100644 --- a/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-RMSE.py +++ b/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-RMSE.py @@ -17,6 +17,7 @@ from deap import algorithms, base, creator, tools from hapi.calibration import Calibration +from hapi.catchment import Catchment from hapi.routing import Routing from hapi.rrm.hbv_bergestrom92 import HBVBergestrom92 as HBVLumped from hapi.run import Run @@ -32,8 +33,8 @@ end = "2011-12-31" name = "Coello" -Coello = Calibration(name, start, end) -Coello.read_lumped_inputs(MeteoDataPath) +Coello = Calibration(Catchment(name, start, end)) +Coello.model.read_lumped_inputs(MeteoDataPath) ### Basic_inputs @@ -42,7 +43,7 @@ # temporal resolution # [Snow pack, Soil moisture, Upper zone, Lower Zone, Water content] InitialCond = [0, 10, 10, 10, 0] -Coello.read_lumped_model(HBVLumped, AreaCoeff, InitialCond) +Coello.model.read_lumped_model(HBVLumped, AreaCoeff, InitialCond) # Calibration parameters @@ -66,7 +67,7 @@ RoutingFn = Routing.triangular_routing_1 # outlet discharge -Coello.read_discharge_gauges(Path + "Qout_c.csv", fmt="%Y-%m-%d") +Coello.model.read_discharge_gauges(Path + "Qout_c.csv", fmt="%Y-%m-%d") # %% Calibration creator.create("Fitness", base.Fitness, weights=(1.0, -1.0)) creator.create("IndividualContainer", list, fitness=creator.Fitness) @@ -95,12 +96,12 @@ def initializer(): def objfn(individual): - # Coello.read_parameters(Parameterpath, Snow) - Coello.parameters = individual + # Coello.model.read_parameters(Parameterpath, Snow) + Coello.model.parameters = individual Run.run_lumped(Coello, Route, RoutingFn) - # [Coello.QGauges.columns[-1]] - NSE = metrics.nse_hf(Coello.QGauges, Coello.Qsim, *Coello.OFArgs) - RMSE = metrics.rmse(Coello.QGauges, Coello.Qsim, *Coello.OFArgs) + # [Coello.model.QGauges.columns[-1]] + NSE = metrics.nse_hf(Coello.model.QGauges, Coello.Qsim, *Coello.OFArgs) + RMSE = metrics.rmse(Coello.model.QGauges, Coello.Qsim, *Coello.OFArgs) return NSE, RMSE @@ -149,7 +150,7 @@ def distance(individual): print("Best individual is %s, %s" % (best_ind, best_ind.fitness.values)) # %% Run the Model -Coello.parameters = best_ind +Coello.model.parameters = best_ind # [0.7686518278956287, 144.35510831203874, 1.9922719933560913, 0.1439126168555068, 0.9474744708723734, # 0.749219030317463, 0.8074091462437563, 0.07289588281400794, 68.83482640397304, 5.123384184968337, # 1.9922719933560913] @@ -159,7 +160,7 @@ def distance(individual): scores = dict() -Qobs = Coello.QGauges[Coello.QGauges.columns[0]] +Qobs = Coello.model.QGauges[Coello.model.QGauges.columns[0]] scores["RMSE"] = metrics.rmse(Qobs, Coello.Qsim["q"]) scores["NSE"] = metrics.nse(Qobs, Coello.Qsim["q"]) @@ -178,7 +179,7 @@ def distance(individual): gaugei = 0 plotstart = "2009-01-01" plotend = "2011-12-31" -Coello.plot_hydrograph(plotstart, plotend, gaugei, title="Lumped Model") +Coello.model.plot_hydrograph(plotstart, plotend, gaugei, title="Lumped Model") # %% Save the Parameters @@ -203,4 +204,4 @@ def distance(individual): + str(dt.datetime.now())[0:10] + ".txt" ) -Coello.save_results(result=5, start=StartDate, end=EndDate, path=Path) +Coello.model.save_results(result=5, start=StartDate, end=EndDate, path=Path) diff --git a/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap.py b/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap.py index 146e1bf3..4ba3560f 100644 --- a/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap.py +++ b/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap.py @@ -15,6 +15,7 @@ from deap import algorithms, base, creator, tools from hapi.calibration import Calibration +from hapi.catchment import Catchment from hapi.routing import Routing from hapi.rrm.hbv_bergestrom92 import HBVBergestrom92 as HBVLumped from hapi.run import Run @@ -29,8 +30,8 @@ end = "2011-12-31" name = "Coello" -Coello = Calibration(name, start, end) -Coello.read_lumped_inputs(MeteoDataPath) +Coello = Calibration(Catchment(name, start, end)) +Coello.model.read_lumped_inputs(MeteoDataPath) # %% Basic_inputs # catchment area AreaCoeff = 1530 @@ -39,7 +40,7 @@ InitialCond = [0, 10, 10, 10, 0] # no snow subroutine Snow = False -Coello.read_lumped_model(HBVLumped, AreaCoeff, InitialCond) +Coello.model.read_lumped_model(HBVLumped, AreaCoeff, InitialCond) # Calibration parameters @@ -61,7 +62,7 @@ RoutingFn = Routing.triangular_routing_1 # outlet discharge -Coello.read_discharge_gauges(Path + "Qout_c.csv", fmt="%Y-%m-%d") +Coello.model.read_discharge_gauges(Path + "Qout_c.csv", fmt="%Y-%m-%d") # %% Calibration creator.create("FitnessMin", base.Fitness, weights=(1.0,)) creator.create("IndividualContainer", list, fitness=creator.FitnessMin) @@ -91,11 +92,11 @@ def initializer(): def objfn(individual): - # Coello.read_parameters(Parameterpath, Snow) - Coello.parameters = individual + # Coello.model.read_parameters(Parameterpath, Snow) + Coello.model.parameters = individual Run.run_lumped(Coello, Route, RoutingFn) - # [Coello.QGauges.columns[-1]] - error = PC.NSEHF(Coello.QGauges, Coello.Qsim, *Coello.OFArgs) + # [Coello.model.QGauges.columns[-1]] + error = PC.NSEHF(Coello.model.QGauges, Coello.Qsim, *Coello.OFArgs) return (error,) @@ -144,7 +145,7 @@ def distance(individual): print("Best individual is %s, %s" % (best_ind, best_ind.fitness.values)) # %% Run the Model -Coello.parameters = best_ind +Coello.model.parameters = best_ind # [0.7686518278956287, 144.35510831203874, 1.9922719933560913, 0.1439126168555068, 0.9474744708723734, # 0.749219030317463, 0.8074091462437563, 0.07289588281400794, 68.83482640397304, 5.123384184968337, # 1.9922719933560913] @@ -154,7 +155,7 @@ def distance(individual): metrics = dict() -Qobs = Coello.QGauges[Coello.QGauges.columns[0]] +Qobs = Coello.model.QGauges[Coello.model.QGauges.columns[0]] metrics["RMSE"] = PC.RMSE(Qobs, Coello.Qsim["q"]) metrics["NSE"] = PC.NSE(Qobs, Coello.Qsim["q"]) @@ -173,7 +174,7 @@ def distance(individual): gaugei = 0 plotstart = "2009-01-01" plotend = "2011-12-31" -Coello.plot_hydrograph(plotstart, plotend, gaugei, title="Lumped Model") +Coello.model.plot_hydrograph(plotstart, plotend, gaugei, title="Lumped Model") # %% Save the Parameters @@ -192,4 +193,4 @@ def distance(individual): Path = ( Path + f"{Coello.name}-results-lumped-model" + str(dt.datetime.now())[0:10] + ".txt" ) -Coello.save_results(result=5, start=StartDate, end=EndDate, path=Path) +Coello.model.save_results(result=5, start=StartDate, end=EndDate, path=Path) diff --git a/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration.py b/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration.py index 81983b9b..5501861c 100644 --- a/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration.py +++ b/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration.py @@ -8,6 +8,7 @@ import statista.descriptors as metrics from hapi.calibration import Calibration +from hapi.catchment import Catchment from hapi.routing import Routing from hapi.rrm.hbv_bergestrom92 import HBVBergestrom92 as HBVLumped from hapi.run import Run @@ -21,8 +22,8 @@ end = "2011-12-31" name = "Coello" -Coello = Calibration(name, start, end) -Coello.read_lumped_inputs(MeteoDataPath) +Coello = Calibration(Catchment(name, start, end)) +Coello.model.read_lumped_inputs(MeteoDataPath) # %% Basic_inputs # catchment area @@ -32,7 +33,7 @@ InitialCond = [0, 10, 10, 10, 0] # no snow subroutine Snow = False -Coello.read_lumped_model(HBVLumped, AreaCoeff, InitialCond) +Coello.model.read_lumped_model(HBVLumped, AreaCoeff, InitialCond) # Calibration parameters @@ -58,7 +59,7 @@ ### Objective function # outlet discharge -Coello.read_discharge_gauges(Path + "Qout_c.csv", fmt="%Y-%m-%d") +Coello.model.read_discharge_gauges(Path + "Qout_c.csv", fmt="%Y-%m-%d") OF_args = [] objective_function = metrics.rmse @@ -106,14 +107,14 @@ print("Time = " + str(round(cal_parameters[2]["time"] / 60, 2)) + " min") # %% Run the Model -Coello.parameters = cal_parameters[1] +Coello.model.parameters = cal_parameters[1] Run.run_lumped(Coello, Route, RoutingFn) ### Calculate Performance Criteria scores = dict() -Qobs = Coello.QGauges[Coello.QGauges.columns[0]] +Qobs = Coello.model.QGauges[Coello.model.QGauges.columns[0]] scores["RMSE"] = metrics.rmse(Qobs, Coello.Qsim["q"]) scores["NSE"] = metrics.nse(Qobs, Coello.Qsim["q"]) @@ -132,7 +133,7 @@ gaugei = 0 plotstart = "2009-01-01" plotend = "2011-12-31" -Coello.plot_hydrograph(plotstart, plotend, gaugei, title="Lumped Model") +Coello.model.plot_hydrograph(plotstart, plotend, gaugei, title="Lumped Model") ### Save the Parameters @@ -147,4 +148,4 @@ EndDate = "2010-04-20" Path = Path + "Results-Lumped-Model" + str(dt.datetime.now())[0:10] + ".txt" -Coello.save_results(result=5, start=StartDate, end=EndDate, path=Path) +Coello.model.save_results(result=5, start=StartDate, end=EndDate, path=Path) diff --git a/examples/hydrological-model/coello/calibration/coello-totally-distributed-model-calibratation-muskingum.py b/examples/hydrological-model/coello/calibration/coello-totally-distributed-model-calibratation-muskingum.py index 6730d836..098f1e4f 100644 --- a/examples/hydrological-model/coello/calibration/coello-totally-distributed-model-calibratation-muskingum.py +++ b/examples/hydrological-model/coello/calibration/coello-totally-distributed-model-calibratation-muskingum.py @@ -5,6 +5,7 @@ from statista.descriptors import rmse from hapi.calibration import Calibration +from hapi.catchment import Catchment from hapi.inputs import FlowNetwork, MeteoInputs from hapi.rrm.hbv_bergestrom92 import HBVBergestrom92 as HBV from hapi.rrm.parameters import Parameters as DP @@ -30,13 +31,15 @@ start_date = "2009-01-01" end_date = "2009-04-10" name = "Coello" -Coello = Calibration(name, start_date, end_date, spatial_resolution="Distributed") +Coello = Calibration( + Catchment(name, start_date, end_date, spatial_resolution="Distributed") +) # %% Meteorological & GIS Data -Coello.meteo = MeteoInputs.from_rasters(PrecPath, TempPath, Evap_Path) +Coello.model.meteo = MeteoInputs.from_rasters(PrecPath, TempPath, Evap_Path) -Coello.flow_network = FlowNetwork.from_rasters(FlowAccPath, FlowDPath) -Coello.read_lumped_model(HBV, AreaCoeff, InitialCond) +Coello.model.flow_network = FlowNetwork.from_rasters(FlowAccPath, FlowDPath) +Coello.model.read_lumped_model(HBV, AreaCoeff, InitialCond) # %% UB = np.loadtxt(CalibPath + "/UB - tot.txt", usecols=0) LB = np.loadtxt(CalibPath + "/LB - tot.txt", usecols=0) @@ -74,13 +77,13 @@ # calculate no of parameters that optimization algorithm is going to generate print(SpatialVarFun.ParametersNO) # %% Gauges -Coello.read_gauge_table(Path + "/stations/gauges.csv", FlowAccPath) +Coello.model.read_gauge_table(Path + "/stations/gauges.csv", FlowAccPath) GaugesPath = Path + "/stations/" -Coello.read_discharge_gauges(GaugesPath, column="id", fmt="%Y-%m-%d") -print(Coello.GaugesTable) +Coello.model.read_discharge_gauges(GaugesPath, column="id", fmt="%Y-%m-%d") +print(Coello.model.GaugesTable) # %% ### Objective function -coordinates = Coello.GaugesTable[["id", "x", "y", "weight"]][:] +coordinates = Coello.model.GaugesTable[["id", "x", "y", "weight"]][:] # define the objective function and its arguments OF_args = [coordinates] @@ -134,5 +137,7 @@ def objective_function(q_obs, coordinates): # %% ### Run Calibration cal_parameters = Coello.run_calibration(SpatialVarFun, OptimizationArgs, print_error=1) # %% -SpatialVarFun.Function(Coello.parameters, kub=SpatialVarFun.Kub, klb=SpatialVarFun.Klb) +SpatialVarFun.Function( + Coello.model.parameters, kub=SpatialVarFun.Kub, klb=SpatialVarFun.Klb +) SpatialVarFun.save_parameters(SaveTo) diff --git a/pyproject.toml b/pyproject.toml index 71850d4e..9e153f1a 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -189,10 +189,15 @@ module = [ "hapi.catchment", "hapi.calibration", ] -# These two are builders: attributes are typed X | None and populated by successive -# read_*() calls, which mypy cannot verify. The run layer no longer needs the excuse -- -# `hapi.runs` narrows a builder into a validated, non-optional run, so `run`, `wrapper` -# and `distrrm` type-check clean. What is left here is the builder side itself. +# The builder side. `Catchment` is assembled by successive read_*() calls, so its inputs +# are X | None until the matching one has run -- which mypy cannot verify and which is +# honest rather than a defect. `Calibration` reads those same optional attributes off the +# model it holds, and its `SpatialVarFun` argument is typed `Callable` though the code uses +# `.Function` / `.Par3d` / `.no_parameters` on it; a Protocol for that would close most of +# the remaining 44. +# +# The run layer no longer needs the excuse: `hapi.runs` narrows a builder into a validated, +# non-optional run, so `run`, `wrapper` and `distrrm` all type-check clean. disable_error_code = [ "union-attr", "attr-defined", diff --git a/src/hapi/calibration.py b/src/hapi/calibration.py index 9093c227..44832bf2 100644 --- a/src/hapi/calibration.py +++ b/src/hapi/calibration.py @@ -12,11 +12,12 @@ from typing import Any import numpy as np +from loguru import logger from Oasis.harmonysearch import HSapi from Oasis.optimization import Optimization from hapi.catchment import Catchment -from hapi.conceptual import ParameterSet +from hapi.conceptual import ParameterBounds, ParameterSet from hapi.runs import DistributedRun, LumpedRun from hapi.wrapper import Wrapper @@ -52,70 +53,98 @@ def _check_optimization_args(api_obj_args: Any, api_solve_args: Any) -> None: ) -class Calibration(Catchment): - """Calibration class for distributed hydrological model parameter optimization. - - The Calibration class connects the parameter spatial distribution function - with both components of the spatial representation of the hydrological - process (conceptual model and spatial routing) to calculate the - performance of predicted runoff at known locations based on a given - performance function. - - The Calibration class is a subclass of the Catchment superclass, so you - need to create the Catchment object first to be able to run the - calibration. - - Note: - Results live on `self.results` (a :class:`~hapi.results.SimulationResults`), so the - objective functions read `self.results.q_total` rather than an attribute on the - catchment. Unlike `Run`, this class is still a `Catchment` subclass; converting it - to composition is tracked separately. +class Calibration: + """Calibrates a catchment's parameters against observed discharge. + + Holds the catchment it calibrates rather than being one. It was a subclass, which meant it + inherited a forty-attribute builder to use a dozen fields of, and inherited + `plot_hydrograph` -- which reads `Qsim.loc[...]` and so could never work against the bare + array this class's own `extract_discharge` produces. Composition removes that class of + problem: nothing is inherited, so nothing can be inherited broken. + + The search space lives here too. `ParameterBounds` is read by nothing else, and it carries + the `(snow, maxbas)` pair every trial vector is checked against, so it belongs beside the + optimiser rather than on the model. + + Attributes: + model: The catchment being calibrated. Build it first, then hand it over. + bounds: The search space, once `read_parameters_bound` has run. + objective_function: The metric being optimised. + OFArgs: Extra arguments forwarded to it. + OFvalue: The best objective value the optimiser found. + best_parameters: The optimiser's answer -- the flat vector it searched over. Not a + runnable parameter set: for a distributed calibration the winning vector still has + to go through the spatial-distribution function to become the `(rows, cols, n)` + array a run reads. The runnable set is `model.parameters`. + Qsim: The simulated hydrograph at the gauge cells, as `extract_discharge` builds it -- + a bare array sized `(time_steps, n_gauges)`, which is what the objective function + consumes. + + Examples: + ```python + >>> from hapi.calibration import Calibration # doctest: +SKIP + >>> from hapi.catchment import Catchment # doctest: +SKIP + >>> model = Catchment.from_yaml("coello.yaml") # doctest: +SKIP + >>> calibration = Calibration(model) # doctest: +SKIP + >>> calibration.read_parameters_bound(upper, lower) # doctest: +SKIP + >>> calibration.read_objective_function(rmse, []) # doctest: +SKIP + ``` """ - def __init__( - self, - name: Any, - start: str, - end: str, - fmt: str = "%Y-%m-%d", - spatial_resolution: str = "Lumped", - temporal_resolution: str = "Daily", - routing_method: str = "Muskingum", - ): - """Initialize the Calibration object. + def __init__(self, model: Catchment): + """Wrap the catchment to be calibrated. Args: - name (Any): Name of the Catchment. - start (str): Starting date as a string. - end (str): End date as a string. - fmt (str, optional): Format of the given date. - Default is "%Y-%m-%d". - spatial_resolution (str, optional): Spatial resolution mode, - either "Lumped" or "Distributed". Default is "Lumped". - temporal_resolution (str, optional): Temporal resolution mode, - either "Hourly" or "Daily". Default is "Daily". - routing_method (str, optional): Routing method name. - Default is "Muskingum". + model: The catchment to calibrate, with its inputs read. Its `parameters` are + replaced once per trial vector, so it comes back carrying the last set tried. + + Raises: + TypeError: `model` is not a `Catchment`. """ - super().__init__( - name, - start, - end, - fmt, - spatial_resolution, - temporal_resolution, - routing_method, - ) + if not isinstance(model, Catchment): + raise TypeError( + f"Calibration takes the Catchment it calibrates, got " + f"{type(model).__name__}; build the model first, then wrap it" + ) + self.model = model + self.bounds: ParameterBounds | None = None self.objective_function: Callable[..., Any] | None = None self.OFArgs: list | None = None self.OFvalue: float | None = None - #: The optimiser's answer -- the flat vector it searched over, as returned in - #: `res[1]`. Deliberately *not* the model's runnable parameter set: for a distributed - #: calibration the winning vector still has to go through the spatial-distribution - #: function to become the `(rows, cols, n)` array a run reads, so the two are - #: different shapes describing different things. The runnable set lives on - #: `self.parameters.values`. self.best_parameters: np.ndarray | list | None = None + self.Qsim: np.ndarray | None = None + + def read_parameters_bound( + self, + upper_bound: list | np.ndarray, + lower_bound: list | np.ndarray, + snow: bool = False, + maxbas: bool = False, + ) -> None: + """Read the search space the optimiser explores. + + Moved here from `Catchment`: nothing but a calibration reads it, and it carries the + `(snow, maxbas)` pair that fixes how wide every trial vector must be -- which is the + rule `_parameter_set` checks each one against. + + Args: + upper_bound: Upper bound per parameter. + lower_bound: Lower bound per parameter. + snow: Whether the snow routine runs. + maxbas: Whether the vector carries a MAXBAS value instead of Muskingum's two. + + Raises: + ValueError: The bounds are different lengths, or `snow` is not a bool. + """ + if not isinstance(snow, bool): + raise ValueError( + "snow input defines whether to consider snow subroutine or not it has to " + "be True or False" + ) + self.bounds = ParameterBounds( + lower_bound, upper_bound, snow=snow, maxbas=maxbas + ) + logger.debug("Parameters' bounds are read successfully") def _declare_the_parameter_variables( self, opt_prob: Optimization, initial_values: list | None = None @@ -169,8 +198,8 @@ def _parameter_set(self, values) -> ParameterSet: Raises: ValueError: The trial set is not the width the configuration requires. """ - if self.parameters is not None: - return self.parameters.with_values(values) + if self.model.parameters is not None: + return self.model.parameters.with_values(values) bounds = self.bounds snow = bounds.snow if bounds is not None else False maxbas = bounds.maxbas if bounds is not None else False @@ -216,7 +245,7 @@ def extract_discharge( """Extract the simulated discharge hydrograph at gauge locations. Extracts discharge values from the total routed discharge array - (`self.results.q_total`) at each gauge location and stores them in + (`self.model.results.q_total`) at each gauge location and stores them in `self.Qsim`. Optionally applies a multiplication factor per gauge. @@ -233,7 +262,7 @@ def extract_discharge( ValueError: The results came from MAXBAS routing, whose per-cell values are contributions rather than discharges. """ - if not self.results.outlet_shortcut_valid: + if not self.model.results.outlet_shortcut_valid: raise ValueError( "this catchment was run with triangular (MAXBAS) routing, which sends " "every cell straight to the outlet: a single cell of q_total is that cell's " @@ -242,19 +271,23 @@ def extract_discharge( "calibrated against the wrong signal." ) - self.Qsim = np.zeros((self.meteo.time_steps, len(self.GaugesTable))) + self.Qsim = np.zeros((self.model.meteo.time_steps, len(self.model.GaugesTable))) # error = 0 - for i in range(len(self.GaugesTable)): - Xind = int(self.GaugesTable.loc[self.GaugesTable.index[i], "cell_row"]) - Yind = int(self.GaugesTable.loc[self.GaugesTable.index[i], "cell_col"]) - # gaugeid = self.GaugesTable.loc[self.GaugesTable.index[i],"id"] + for i in range(len(self.model.GaugesTable)): + Xind = int( + self.model.GaugesTable.loc[self.model.GaugesTable.index[i], "cell_row"] + ) + Yind = int( + self.model.GaugesTable.loc[self.model.GaugesTable.index[i], "cell_col"] + ) + # gaugeid = self.model.GaugesTable.loc[self.model.GaugesTable.index[i],"id"] - # Quz = self.results.quz_routed[Xind,Yind,:-1] - # Qlz = self.results.qlz_translated[Xind,Yind,:-1] + # Quz = self.model.results.quz_routed[Xind,Yind,:-1] + # Qlz = self.model.results.qlz_translated[Xind,Yind,:-1] # self.Qsim[:,i] = Quz + Qlz Qsim = np.reshape( - self.results.q_total[Xind, Yind, :-1], self.meteo.time_steps + self.model.results.q_total[Xind, Yind, :-1], self.model.meteo.time_steps ) if factor is not None: @@ -320,14 +353,19 @@ def run_calibration( """ # input dimensions # [rows,cols] = self.FlowAcc.ReadAsArray().shape - [fd_rows, fd_cols] = self.flow_network.flow_dir_arr.shape - if fd_rows != self.flow_network.rows or fd_cols != self.flow_network.cols: + [fd_rows, fd_cols] = self.model.flow_network.flow_dir_arr.shape + if ( + fd_rows != self.model.flow_network.rows + or fd_cols != self.model.flow_network.cols + ): raise ValueError(ROWS_MISMATCH_ERROR) # The three cubes already agree with each other (checked when MeteoInputs was # built); this is the other half -- that they cover the model's grid. - self.meteo.validate_against( - self.flow_network.rows, self.flow_network.cols, self.period.date_index + self.model.meteo.validate_against( + self.model.flow_network.rows, + self.model.flow_network.cols, + self.model.period.date_index, ) # basic inputs @@ -348,31 +386,28 @@ def run_calibration( ### calculate the objective function def opt_fun(par): + # Distributing the parameters and narrowing the model both happen *outside* the + # try. They are checks on the setup, not on this candidate: a wrong-width vector or + # a grid mismatch is a bug to surface, and scoring it `nan` would let the optimiser + # search on over a model that never ran -- which is what the bare `except` did. + spatial_var_fun.Function(par) + self.model.parameters = self._parameter_set(spatial_var_fun.Par3d) + run = DistributedRun.from_model(self.model) + try: - # distribute the parameters - spatial_var_fun.Function( - par - ) # , kub=spatial_var_fun.Kub, klb=spatial_var_fun.Klb - # Re-checked per trial: `with_parameters` re-runs the (snow, maxbas) count - # rule, so a distribution function producing the wrong width fails here - # rather than as an index error inside the per-cell loop. - self.parameters = self._parameter_set(spatial_var_fun.Par3d) - # Narrowing is the validation: every trial vector is checked against the grid - # and the period here, on the one path that used to skip every check `Run` - # made by going straight to the wrapper. - self.results = Wrapper.run_muskingum(DistributedRun.from_model(self)) + self.model.results = Wrapper.run_muskingum(run) # calculate performance of the model try: error = self.objective_function( - self.QGauges, *[self.GaugesTable] - ) # self.results.qout, self.results.quz_routed, self.results.qlz_translated, + self.model.QGauges, *[self.model.GaugesTable] + ) # self.model.results.qout, self.model.results.quz_routed, self.model.results.qlz_translated, f = list(range(9, len(par), spatial_var_fun.no_parameters)) g = list() for i in range(len(f)): k = par[f[i]] x = par[f[i] + 1] - g.append(2 * k * x / self.period.dt) - g.append((2 * k * (1 - x)) / self.period.dt) + g.append(2 * k * x / self.model.period.dt) + g.append((2 * k * (1 - x)) / self.model.period.dt) except TypeError as e: # the objective function received fewer inputs than it needs @@ -384,7 +419,11 @@ def opt_fun(par): print(par) fail = 0 - except: + except Exception as exc: + # A genuine numerical failure for this candidate. Narrowed from a bare + # `except`, which also caught KeyboardInterrupt -- so a long calibration + # could not be stopped -- and reported every defect as a bad parameter set. + logger.warning(f"trial failed, scoring it infeasible: {exc!r}") error = np.nan g = [] fail = 1 @@ -477,8 +516,10 @@ def calibrate_maxbas( # The three cubes already agree with each other (checked when MeteoInputs was # built); this is the other half -- that they cover the model's grid. - self.meteo.validate_against( - self.flow_network.rows, self.flow_network.cols, self.period.date_index + self.model.meteo.validate_against( + self.model.flow_network.rows, + self.model.flow_network.cols, + self.model.period.date_index, ) # basic inputs @@ -499,21 +540,20 @@ def calibrate_maxbas( # calculate the objective function def opt_fun(par): + # See run_calibration: the setup checks belong outside the try, so a wrong-width + # vector or a grid mismatch surfaces instead of being scored `nan`. + spatial_var_fun.Function(par) + self.model.parameters = self._parameter_set(spatial_var_fun.Par3d) + run = DistributedRun.from_model(self.model, needs_flow_direction=False) + try: - # distribute the parameters - spatial_var_fun.Function( - par - ) # , kub=spatial_var_fun.Kub, klb=spatial_var_fun.Klb, Maskingum=spatial_var_fun.Maskingum - # Re-checked per trial -- see run_calibration. - self.parameters = self._parameter_set(spatial_var_fun.Par3d) - # See run_calibration: narrowing validates each trial vector. - self.results = Wrapper.run_maxbas( - DistributedRun.from_model(self, needs_flow_direction=False) - ) + self.model.results = Wrapper.run_maxbas(run) # calculate performance of the model try: error = self.objective_function( - self.QGauges, self.results.qout, *[self.GaugesTable] + self.model.QGauges, + self.model.results.qout, + *[self.model.GaugesTable], ) except TypeError as e: # the objective function received fewer inputs than it needs @@ -525,7 +565,9 @@ def opt_fun(par): print(par) fail = 0 - except: + except Exception as exc: + # See run_calibration: narrowed from a bare `except`. + logger.warning(f"trial failed, scoring it infeasible: {exc!r}") error = np.nan fail = 1 @@ -639,22 +681,23 @@ def calibrate_lumped( ### calculate the objective function def opt_fun(par): + # See run_calibration: the setup checks belong outside the try. + self.model.parameters = self._parameter_set(par) + run = LumpedRun.from_model(self.model) + try: - # parameters. Checked against (snow, maxbas) as it arrives. - self.parameters = self._parameter_set(par) - # See run_calibration: narrowing validates each trial vector. - self.results = Wrapper.run_lumped( - LumpedRun.from_model(self), route, routing_fn - ) - self.Qsim = self.results.q_total + self.model.results = Wrapper.run_lumped(run, route, routing_fn) + self.Qsim = self.model.results.q_total # calculate performance of the model try: error = self.objective_function( - self.QGauges[self.QGauges.columns[-1]], self.Qsim, *self.OFArgs + self.model.QGauges[self.model.QGauges.columns[-1]], + self.Qsim, + *self.OFArgs, ) g = [ - 2 * par[-2] * par[-1] / self.period.dt, - (2 * par[-2] * (1 - par[-1])) / self.period.dt, + 2 * par[-2] * par[-1] / self.model.period.dt, + (2 * par[-2] * (1 - par[-1])) / self.model.period.dt, ] except TypeError as e: # the objective function received fewer inputs than it needs @@ -666,7 +709,11 @@ def opt_fun(par): ) # print(par) fail = 0 - except: + except Exception as exc: + # A genuine numerical failure for this candidate. Narrowed from a bare + # `except`, which also caught KeyboardInterrupt -- so a long calibration + # could not be stopped -- and reported every defect as a bad parameter set. + logger.warning(f"trial failed, scoring it infeasible: {exc!r}") error = np.nan g = [] fail = 1 diff --git a/src/hapi/catchment.py b/src/hapi/catchment.py index 74d8b9d8..792ef437 100644 --- a/src/hapi/catchment.py +++ b/src/hapi/catchment.py @@ -35,7 +35,7 @@ from pyramids.dataset import DatasetCollection as Datacube from pyramids.feature import FeatureCollection -from hapi.conceptual import ConceptualModelSetup, ParameterBounds, ParameterSet +from hapi.conceptual import ConceptualModelSetup, ParameterSet from hapi.config import RunConfig from hapi.inputs import ( METEO_VARIABLES, @@ -323,8 +323,6 @@ def __init__( self.meteo: MeteoInputs | None = None self.QGauges: pd.DataFrame | None = None self.GaugesTable: FeatureCollection | pd.DataFrame | None = None - #: The search space a calibration explores, once read. `None` otherwise. - self.bounds: ParameterBounds | None = None #: The routing network and the grid it defines. Assign a #: :class:`~hapi.inputs.FlowNetwork` built by its loader. self.flow_network: FlowNetwork | None = None @@ -360,9 +358,9 @@ def from_yaml(cls, path: str | Path) -> Self: Running the model stays the caller's job, through whichever `Run.*` entry point suits `routing_method` and `spatial_resolution`. - Builds `cls`, so `Calibration.from_yaml(...)` returns a `Calibration` -- it takes the - same constructor arguments. `Run` is not a catchment at all and has nothing to build; - `Run.from_yaml` exists only to say so and point here. + Neither `Run` nor `Calibration` is a catchment any more, so there is no subclass for + this to build: `Run` is a namespace of entry points, and `Calibration` takes the model + it calibrates -- `Calibration(Catchment.from_yaml(path))`. Args: path: Path to the YAML file, as a string or a `Path`. See :mod:`hapi.config` for @@ -430,10 +428,6 @@ def from_yaml(cls, path: str | Path) -> Self: _resolve_config_paths(config, Path(path).resolve().parent) catchment = config.catchment - # The first three go positionally on purpose: `Catchment.__init__` calls its second - # parameter `start_data` and `Calibration.__init__` calls it `start`, so naming them - # would break `Calibration.from_yaml` -- which this method is documented to support -- - # while still working here. Renaming the parameter is the fix, and is breaking. model = cls( catchment.name, catchment.start, @@ -1042,43 +1036,6 @@ def _read_the_single_discharge_file( self.period.start : self.period.end, f.columns[0] ] - def read_parameters_bound( - self, - upper_bound: list | np.ndarray, - lower_bound: list | np.ndarray, - snow: bool = False, - maxbas: bool = False, - ): - """Read the lower and upper parameter bounds for calibration. - - Args: - upper_bound (list | np.ndarray): Upper bound values - for each parameter. - lower_bound (list | np.ndarray): Lower bound values - for each parameter. - snow (bool, optional): Whether to simulate snow - processes. If True, snow-related parameters must be - bounded. Default is False. - maxbas (bool, optional): True if the parameters include - maxbas. Default is False. - - Raises: - ValueError: If the lengths of `upper_bound` and - `lower_bound` are not equal. - ValueError: If `snow` is not a boolean. - """ - if not isinstance(snow, bool): - raise ValueError( - " snow input defines whether to consider snow subroutine or not it has to be True or False" - ) - # A calibration reads no parameter file, so the bounds are where `(snow, maxbas)` - # enters -- carried here so every trial vector can be checked against it. - self.bounds = ParameterBounds( - lower_bound, upper_bound, snow=snow, maxbas=maxbas - ) - - logger.debug("Parameters' bounds are read successfully") - def extract_discharge(self, calculate_metrics=True, factor=None): """Extract and sum discharge at gauge locations. diff --git a/src/hapi/runs.py b/src/hapi/runs.py index 3f98eaa5..95bc47c3 100644 --- a/src/hapi/runs.py +++ b/src/hapi/runs.py @@ -56,7 +56,9 @@ def _require(model: CatchmentLike, name: str, hint: str) -> Any: """ value = getattr(model, name, None) if value is None: - raise ValueError(f"this run needs {name}, which is not set on the model; {hint}") + raise ValueError( + f"this run needs {name}, which is not set on the model; {hint}" + ) return value @@ -108,7 +110,9 @@ def __post_init__(self): if shape[1] != cols: raise ValueError(COLS_MISMATCH_ERROR) - if self.river_geometry is not None and not self.river_geometry.covers(rows, cols): + if self.river_geometry is not None and not self.river_geometry.covers( + rows, cols + ): raise ValueError(GRID_MISMATCH_ERROR) if self.skip_hydraulic_cells and self.river_geometry is None: @@ -191,7 +195,10 @@ def from_model( # `FlowNetwork.__post_init__` already checks this at construction, but # `__setattr__` does not re-check on replacement (unlike `MeteoInputs`), so a # raster swapped in afterwards can still disagree with the grid. - if flow_network.flow_dir_arr.shape != (flow_network.rows, flow_network.cols): + if flow_network.flow_dir_arr.shape != ( + flow_network.rows, + flow_network.cols, + ): raise ValueError(GRID_MISMATCH_ERROR) if flow_network.FDT is None: raise ValueError( @@ -201,7 +208,9 @@ def from_model( geometry = None if with_river_geometry or skip_hydraulic_cells: - geometry = _require(model, "river_geometry", "call read_river_geometry first") + geometry = _require( + model, "river_geometry", "call read_river_geometry first" + ) return cls( period=model.period, diff --git a/tests/calibration/distributed_mode_calib.py b/tests/calibration/distributed_mode_calib.py index 77e33431..cdb2c058 100644 --- a/tests/calibration/distributed_mode_calib.py +++ b/tests/calibration/distributed_mode_calib.py @@ -12,6 +12,7 @@ from pyramids.dataset import Dataset from hapi.calibration import Calibration +from hapi.catchment import Catchment from hapi.inputs import FlowNetwork, MeteoInputs from hapi.rrm.hbv_bergestrom92 import HBVBergestrom92 as HBV from hapi.rrm.parameters import Parameters as DP @@ -37,12 +38,12 @@ Sdate = "2009-01-01" Edate = "2011-12-31" name = "Coello" -Coello = Calibration(name, Sdate, Edate, spatial_resolution="Distributed") +Coello = Calibration(Catchment(name, Sdate, Edate, spatial_resolution="Distributed")) ### Meteorological & GIS Data -Coello.meteo = MeteoInputs.from_rasters(PrecPath, TempPath, Evap_Path) +Coello.model.meteo = MeteoInputs.from_rasters(PrecPath, TempPath, Evap_Path) -Coello.flow_network = FlowNetwork.from_rasters(FlowAccPath, FlowDPath) -Coello.read_lumped_model(HBV, AreaCoeff, InitialCond) +Coello.model.flow_network = FlowNetwork.from_rasters(FlowAccPath, FlowDPath) +Coello.model.read_lumped_model(HBV, AreaCoeff, InitialCond) UB = np.loadtxt(CalibPath + "/UB - tot.txt", usecols=0) @@ -81,11 +82,11 @@ # calculate no of parameters that optimization algorithm is going to generate spatial_var_fun.ParametersNO # %% Gauges -Coello.read_gauge_table(Path + "/stations/gauges.csv", FlowAccPath) +Coello.model.read_gauge_table(Path + "/stations/gauges.csv", FlowAccPath) GaugesPath = Path + "/stations/" -Coello.read_discharge_gauges(GaugesPath, column="id", fmt="%Y-%m-%d") +Coello.model.read_discharge_gauges(GaugesPath, column="id", fmt="%Y-%m-%d") # %% Objective function -coordinates = Coello.GaugesTable[["id", "x", "y", "weight"]][:] +coordinates = Coello.model.GaugesTable[["id", "x", "y", "weight"]][:] # define the objective function and its arguments OF_args = [coordinates] @@ -138,8 +139,8 @@ def objective_function(Qobs, Qout, q_uz_routed, q_lz_trans, coordinates): spatial_var_fun, optimization_args, print_error=0 ) # %% convert parameters to rasters -# Coello.parameters.values = [0.700, 399, 1.704, 0.1021, 0.4622, 0.6237, 0.1251, 0.005, 59.85, 5.241, 94.91, 0.2075] +# Coello.model.parameters.values = [0.700, 399, 1.704, 0.1021, 0.4622, 0.6237, 0.1251, 0.005, 59.85, 5.241, 94.91, 0.2075] spatial_var_fun.Function( - Coello.parameters.values, kub=spatial_var_fun.Kub, klb=spatial_var_fun.Klb + Coello.model.parameters.values, kub=spatial_var_fun.Kub, klb=spatial_var_fun.Klb ) spatial_var_fun.save_parameters(SaveTo) diff --git a/tests/calibration/lumped_calibration.py b/tests/calibration/lumped_calibration.py index 4d9242eb..c4ca5c4b 100644 --- a/tests/calibration/lumped_calibration.py +++ b/tests/calibration/lumped_calibration.py @@ -5,6 +5,7 @@ import statista.descriptors as metrics from hapi.calibration import Calibration +from hapi.catchment import Catchment from hapi.routing import Routing from hapi.rrm.hbv_bergestrom92 import HBVBergestrom92 as HBVLumped from hapi.run import Run @@ -18,8 +19,8 @@ end = "2011-12-31" name = "Coello" -Coello = Calibration(name, start, end) -Coello.read_lumped_inputs(MeteoDataPath) +Coello = Calibration(Catchment(name, start, end)) +Coello.model.read_lumped_inputs(MeteoDataPath) # %% basic_inputs # catchment area AreaCoeff = 1530 @@ -28,7 +29,7 @@ InitialCond = [0, 10, 10, 10, 0] # no snow subroutine Snow = 0 -Coello.read_lumped_model(HBVLumped, AreaCoeff, InitialCond) +Coello.model.read_lumped_model(HBVLumped, AreaCoeff, InitialCond) # %% Calibration parameters # Calibration boundaries UB = pd.read_csv(Path + "/UB-3.txt", index_col=0, header=None) @@ -49,7 +50,7 @@ # %% ### Objective function # outlet discharge -Coello.read_discharge_gauges(Path + "Qout_c.csv", fmt="%Y-%m-%d") +Coello.model.read_discharge_gauges(Path + "Qout_c.csv", fmt="%Y-%m-%d") OF_args = [] objective_function = metrics.rmse @@ -95,12 +96,12 @@ print("Parameters are " + str(cal_parameters[1])) print("Time = " + str(round(cal_parameters[2]["time"] / 60, 2)) + " min") # %% run the model -Coello.parameters.values = cal_parameters[1] +Coello.model.parameters.values = cal_parameters[1] Run.run_lumped(Coello, Route, routing_fn) # %% calculate performance criteria scores = dict() -Qobs = Coello.QGauges[Coello.QGauges.columns[0]] +Qobs = Coello.model.QGauges[Coello.model.QGauges.columns[0]] scores["RMSE"] = metrics.rmse(Qobs, Coello.Qsim["q"]) scores["NSE"] = metrics.nse(Qobs, Coello.Qsim["q"]) @@ -117,7 +118,7 @@ gaugei = 0 plotstart = "2009-01-01" plotend = "2011-12-31" -Coello.plot_hydrograph(plotstart, plotend, gaugei, title="Lumped Model") +Coello.model.plot_hydrograph(plotstart, plotend, gaugei, title="Lumped Model") # %% save the parameters ParPath = Path + "Parameters" + str(dt.datetime.now())[0:10] + ".txt" @@ -130,4 +131,4 @@ EndDate = "2010-04-20" Path = Path + "Results-Lumped-Model" + str(dt.datetime.now())[0:10] + ".txt" -Coello.save_results(result=5, StartDate=StartDate, EndDate=EndDate, path=Path) +Coello.model.save_results(result=5, StartDate=StartDate, EndDate=EndDate, path=Path) diff --git a/tests/rrm/calibration/test_calibration_distributed.py b/tests/rrm/calibration/test_calibration_distributed.py index 57be2bc8..a069431e 100644 --- a/tests/rrm/calibration/test_calibration_distributed.py +++ b/tests/rrm/calibration/test_calibration_distributed.py @@ -14,6 +14,7 @@ from hapi import calibration as calibration_module from hapi.calibration import Calibration +from hapi.catchment import Catchment from hapi.conceptual import ParameterBounds from hapi.inputs import FlowNetwork, MeteoInputs from hapi.results import RoutingKind, SimulationResults @@ -73,13 +74,15 @@ def gauged_calibration( synthetic `q_total` field, with no model run behind it. """ coello = Calibration( - "coello", - coello_start_date, - coello_end_date, - spatial_resolution="Distributed", - temporal_resolution="Daily", + Catchment( + "coello", + coello_start_date, + coello_end_date, + spatial_resolution="Distributed", + temporal_resolution="Daily", + ) ) - coello.meteo = MeteoInputs.from_rasters( + coello.model.meteo = MeteoInputs.from_rasters( coello_prec_path, coello_temp_path, coello_evap_path, @@ -89,29 +92,31 @@ def gauged_calibration( date=True, file_name_data_fmt="%Y.%m.%d", ) - coello.flow_network = FlowNetwork.from_rasters(coello_acc_path, coello_fd_path) + coello.model.flow_network = FlowNetwork.from_rasters( + coello_acc_path, coello_fd_path + ) # Needed even though the objective overwrites `parameters`: `read_parameters` is also # what sets `snow`, and HBV's parameter parse branches on it being exactly 0 or 1. Left # as None it raises inside the run, which the objective's bare `except` swallows. - coello.read_parameters(coello_dist_parameters_muskingum, False) - coello.read_lumped_model(HBVLumped, coello_cat_area, coello_initial_cond) - coello.GaugesTable = DataFrame( + coello.model.read_parameters(coello_dist_parameters_muskingum, False) + coello.model.read_lumped_model(HBVLumped, coello_cat_area, coello_initial_cond) + coello.model.GaugesTable = DataFrame( {"id": [1, 2], "cell_row": [2, 5], "cell_col": [3, 6]} ) - rows, cols = coello.flow_network.rows, coello.flow_network.cols - steps = coello.meteo.time_steps + rows, cols = coello.model.flow_network.rows, coello.model.flow_network.cols + steps = coello.model.meteo.time_steps rng = np.random.default_rng(1337) # Stage the post-run state the way the run layer builds it. `q_total` and the rest are # read-only views onto `results`, so a finished Muskingum run is described rather than # poked in field by field. - coello.results = SimulationResults( + coello.model.results = SimulationResults( routing=RoutingKind.MUSKINGUM, quz=np.zeros((rows, cols, steps + 1)), qlz=np.zeros((rows, cols, steps + 1)), state_variables=np.zeros((rows, cols, steps + 1, 5)), q_total=rng.random((rows, cols, steps + 1)), ) - coello.QGauges = DataFrame(rng.random((steps, 2)), columns=[1, 2]) + coello.model.QGauges = DataFrame(rng.random((steps, 2)), columns=[1, 2]) return coello @@ -132,18 +137,18 @@ def test_fills_qsim_from_qtot_at_each_gauge_cell( coello.extract_discharge() - expected_shape = (coello.meteo.time_steps, 2) + expected_shape = (coello.model.meteo.time_steps, 2) assert coello.Qsim.shape == expected_shape, ( f"Expected Qsim shape {expected_shape}, got {coello.Qsim.shape}" ) np.testing.assert_allclose( coello.Qsim[:, 0], - coello.results.q_total[2, 3, :-1], + coello.model.results.q_total[2, 3, :-1], err_msg="gauge 1 must come from cell (2, 3) of q_total", ) np.testing.assert_allclose( coello.Qsim[:, 1], - coello.results.q_total[5, 6, :-1], + coello.model.results.q_total[5, 6, :-1], err_msg="gauge 2 must come from cell (5, 6) of q_total", ) @@ -162,12 +167,12 @@ def test_factor_scales_each_gauge_independently( np.testing.assert_allclose( coello.Qsim[:, 0], - coello.results.q_total[2, 3, :-1] * 2.0, + coello.model.results.q_total[2, 3, :-1] * 2.0, err_msg="gauge 1 must be scaled by its own factor", ) np.testing.assert_allclose( coello.Qsim[:, 1], - coello.results.q_total[5, 6, :-1] * 10.0, + coello.model.results.q_total[5, 6, :-1] * 10.0, err_msg="gauge 2 must be scaled by its own factor", ) @@ -182,7 +187,7 @@ def test_rejects_a_catchment_routed_with_maxbas( it would fit the wrong signal, so the guard must refuse rather than return numbers. """ coello = gauged_calibration - coello.results.routing = RoutingKind.MAXBAS + coello.model.results.routing = RoutingKind.MAXBAS with pytest.raises(ValueError, match="MAXBAS") as exc_info: coello.extract_discharge() @@ -238,10 +243,10 @@ def test_rejects_meteo_that_does_not_cover_the_grid( coello.bounds = ParameterBounds(np.zeros(12), np.ones(12)) # Built in one go: replacing the cubes one at a time is now refused, because a # half-applied crop is exactly the inconsistency MeteoInputs guarantees against. - coello.meteo = MeteoInputs( - precipitation=coello.meteo.precipitation[:, :-1, :], - temperature=coello.meteo.temperature[:, :-1, :], - evapotranspiration=coello.meteo.evapotranspiration[:, :-1, :], + coello.model.meteo = MeteoInputs( + precipitation=coello.model.meteo.precipitation[:, :-1, :], + temperature=coello.model.meteo.temperature[:, :-1, :], + evapotranspiration=coello.model.meteo.evapotranspiration[:, :-1, :], ) with pytest.raises(ValueError, match="must share the catchment's grid"): @@ -411,8 +416,11 @@ def test_stores_the_optimizer_result_on_the_instance( ones, so it is pinned separately: `OFvalue` is still res[0] and `parameters` still res[1]. """ - coello = Calibration("rrm", coello_rrm_date[0], coello_rrm_date[1]) - coello.read_lumped_inputs(lumped_meteo_data_path) + coello = Calibration(Catchment("rrm", coello_rrm_date[0], coello_rrm_date[1])) + coello.model.read_lumped_inputs(lumped_meteo_data_path) + # Without this the trials failed and the bare `except` scored them all `nan`; the stub + # returns its canned result regardless, so the test passed over a model that never ran. + coello.model.read_lumped_model(HBVLumped, 1530, [0, 10, 10, 10, 0]) coello.bounds = ParameterBounds(np.zeros(12), np.ones(12)) basic_inputs = dict( Route=0, RoutingFn=Routing.triangular_routing_1, InitialValues=[] @@ -446,8 +454,11 @@ def test_initial_values_are_seeded_into_the_problem( Both branches must declare one variable per bound, so the problem is the same size whether or not a warm start was given. """ - coello = Calibration("rrm", coello_rrm_date[0], coello_rrm_date[1]) - coello.read_lumped_inputs(lumped_meteo_data_path) + coello = Calibration(Catchment("rrm", coello_rrm_date[0], coello_rrm_date[1])) + coello.model.read_lumped_inputs(lumped_meteo_data_path) + # Without this the trials failed and the bare `except` scored them all `nan`; the stub + # returns its canned result regardless, so the test passed over a model that never ran. + coello.model.read_lumped_model(HBVLumped, 1530, [0, 10, 10, 10, 0]) coello.bounds = ParameterBounds(np.zeros(12), np.ones(12)) basic_inputs = dict( Route=0, @@ -480,8 +491,11 @@ def test_a_mismatched_initial_values_length_is_refused( leaving `opt_prob` half-populated and raising `IndexError` far from the call that supplied the list. The length is now compared up front. """ - coello = Calibration("rrm", coello_rrm_date[0], coello_rrm_date[1]) - coello.read_lumped_inputs(lumped_meteo_data_path) + coello = Calibration(Catchment("rrm", coello_rrm_date[0], coello_rrm_date[1])) + coello.model.read_lumped_inputs(lumped_meteo_data_path) + # Without this the trials failed and the bare `except` scored them all `nan`; the stub + # returns its canned result regardless, so the test passed over a model that never ran. + coello.model.read_lumped_model(HBVLumped, 1530, [0, 10, 10, 10, 0]) coello.bounds = ParameterBounds(np.zeros(12), np.ones(12)) basic_inputs = dict( Route=0, @@ -562,7 +576,8 @@ def spatial_var_stub(gauged_calibration: Calibration) -> _SpatialVarStub: _SpatialVarStub: Object exposing `Function`, `Par3d`, `no_parameters`, `no_elem`. """ return _SpatialVarStub( - gauged_calibration.flow_network.rows, gauged_calibration.flow_network.cols + gauged_calibration.model.flow_network.rows, + gauged_calibration.model.flow_network.cols, ) diff --git a/tests/rrm/calibration/test_rrm_calibration.py b/tests/rrm/calibration/test_rrm_calibration.py index f53dd3f0..fc5423ac 100644 --- a/tests/rrm/calibration/test_rrm_calibration.py +++ b/tests/rrm/calibration/test_rrm_calibration.py @@ -1,9 +1,11 @@ import datetime as dt import numpy as np +import pytest import statista.descriptors as metrics from hapi.calibration import Calibration +from hapi.catchment import Catchment from hapi.routing import Routing from hapi.rrm.hbv_bergestrom92 import HBVBergestrom92 as HBVLumped @@ -13,7 +15,7 @@ def test_read_parameters_bounds( lower_bound: list, upper_bound: list, ): - Coello = Calibration("rrm", coello_rrm_date[0], coello_rrm_date[1]) + Coello = Calibration(Catchment("rrm", coello_rrm_date[0], coello_rrm_date[1])) Maxbas = True Snow = False Coello.read_parameters_bound(lower_bound, upper_bound, Snow, maxbas=Maxbas) @@ -36,9 +38,9 @@ def test_lumped_calibration( coello_gauges_date_fmt: str, history_files: str, ): - Coello = Calibration("rrm", coello_rrm_date[0], coello_rrm_date[1]) - Coello.read_lumped_inputs(lumped_meteo_data_path) - Coello.read_lumped_model(HBVLumped, coello_AreaCoeff, coello_InitialCond) + Coello = Calibration(Catchment("rrm", coello_rrm_date[0], coello_rrm_date[1])) + Coello.model.read_lumped_inputs(lumped_meteo_data_path) + Coello.model.read_lumped_model(HBVLumped, coello_AreaCoeff, coello_InitialCond) Maxbas = True Coello.read_parameters_bound(lower_bound, upper_bound, coello_Snow, maxbas=Maxbas) @@ -50,7 +52,7 @@ def test_lumped_calibration( basic_inputs = dict(Route=Route, RoutingFn=routing_fn, InitialValues=parameters) # discharge gauges - Coello.read_discharge_gauges(lumped_gauges_path, fmt=coello_gauges_date_fmt) + Coello.model.read_discharge_gauges(lumped_gauges_path, fmt=coello_gauges_date_fmt) OF_args = [] objective_function = metrics.rmse @@ -89,23 +91,77 @@ def test_create_calibration_instance( self, coello_start_date: str, coello_end_date: str ): coello = Calibration( - "coello", - coello_start_date, - coello_end_date, - spatial_resolution="Distributed", - temporal_resolution="Daily", - fmt="%Y-%m-%d", + Catchment( + "coello", + coello_start_date, + coello_end_date, + spatial_resolution="Distributed", + temporal_resolution="Daily", + fmt="%Y-%m-%d", + ) ) - assert coello.spatial_resolution == "distributed" - assert coello.routing_method == "Muskingum" - assert isinstance(coello.period.start, dt.datetime) + assert coello.model.spatial_resolution == "distributed" + assert coello.model.routing_method == "Muskingum" + assert isinstance(coello.model.period.start, dt.datetime) def test_read_objective_fn(self, coello_start_date: str, coello_end_date: str): coello = Calibration( - "coello", - coello_start_date, - coello_end_date, + Catchment( + "coello", + coello_start_date, + coello_end_date, + ) ) coello.read_objective_function(metrics.rmse, []) assert coello.objective_function == metrics.rmse assert coello.OFArgs == [] + + +class TestCalibrationHoldsACatchment: + """`Calibration` composes a catchment rather than being one.""" + + def test_it_is_not_a_catchment_subclass(self): + """Test that the inheritance is gone. + + Test scenario: + It inherited a forty-attribute builder to use a dozen fields of, and inherited + `plot_hydrograph` -- which reads `Qsim.loc[...]` and so could never work against + the bare array this class's own `extract_discharge` produces. Composition makes + that impossible rather than merely fixed. + """ + assert not issubclass(Calibration, Catchment), ( + "Calibration must hold a Catchment, not be one" + ) + assert not hasattr(Calibration, "plot_hydrograph"), ( + "it must not inherit a plotting method its own extract_discharge would break" + ) + + def test_it_refuses_anything_but_a_catchment(self): + """Test that the constructor takes the model it calibrates. + + Test scenario: + The old signature mirrored `Catchment.__init__` and built one internally, so a + caller could not calibrate a model they had already assembled -- notably one from + `Catchment.from_yaml`. + """ + with pytest.raises(TypeError, match="takes the Catchment it calibrates"): + Calibration("coello") + + def test_the_bounds_live_on_the_calibration(self, coello_rrm_date: list): + """Test that the search space belongs to the optimiser, not the model. + + Test scenario: + `ParameterBounds` is read by nothing but a calibration, and carries the + `(snow, maxbas)` pair every trial vector is checked against -- so keeping it on + `Catchment` put calibration configuration on the model being calibrated. + """ + model = Catchment("rrm", coello_rrm_date[0], coello_rrm_date[1]) + calibration = Calibration(model) + + assert not hasattr(model, "read_parameters_bound"), ( + "read_parameters_bound must move to Calibration with the bounds it builds" + ) + calibration.read_parameters_bound([1.0] * 12, [0.0] * 12) + + assert len(calibration.bounds) == 12, "the bounds land on the calibration" + assert not hasattr(model, "bounds"), "and not on the model" diff --git a/tests/rrm/catchment/test_config.py b/tests/rrm/catchment/test_config.py index 4cb6565e..d8a7fe7f 100644 --- a/tests/rrm/catchment/test_config.py +++ b/tests/rrm/catchment/test_config.py @@ -1288,20 +1288,6 @@ def test_a_mode_argument_that_is_not_a_string_names_itself(self, argument): f"the error should name what it got: {exc.value}" ) - def test_calibration_accepts_the_same_three_arguments(self): - """Test that the guard reaches `Calibration`, which passes the arguments down. - - Test scenario: - `Calibration.__init__` forwards all three to `Catchment.__init__` unchanged, and - its annotations used to advertise a `None` that crashed. - """ - with pytest.raises(TypeError, match="routing_method must be a string"): - Calibration("Coello", "2009-01-01", "2009-01-10", routing_method=None) - - -class TestCatchmentFromYaml: - """Tests for `Catchment.from_yaml`, which turns a configuration into a built model.""" - def test_a_distributed_configuration_populates_every_input( self, distributed_mapping, tmp_path, coello_cat_area, coello_initial_cond ): @@ -1492,7 +1478,7 @@ def test_an_unregistered_model_class_is_refused_before_the_readers_run( with pytest.raises(ValueError, match="not.*registered"): Catchment.from_yaml(path) - @pytest.mark.parametrize("cls", [Catchment, Calibration]) + @pytest.mark.parametrize("cls", [Catchment]) def test_the_builder_returns_the_class_it_was_called_on( self, distributed_mapping, tmp_path, cls ): diff --git a/tests/rrm/catchment/test_rrm_catchment.py b/tests/rrm/catchment/test_rrm_catchment.py index 8b202d60..3b6cc7dd 100644 --- a/tests/rrm/catchment/test_rrm_catchment.py +++ b/tests/rrm/catchment/test_rrm_catchment.py @@ -227,29 +227,6 @@ def test_read_lumped_model( assert coello.model_setup.area == coello_cat_area assert coello.model_setup.initial_cond == coello_initial_cond - def test_read_parameters_bound( - self, - coello_start_date: str, - coello_end_date: str, - coello_parameter_bounds: Tuple[List, List], - ): - LB = coello_parameter_bounds[0] - UB = coello_parameter_bounds[1] - Snow = False - coello = Catchment( - "coello", - coello_start_date, - coello_end_date, - spatial_resolution="Distributed", - temporal_resolution="Daily", - fmt="%Y-%m-%d", - ) - coello.read_parameters_bound(UB, LB, Snow) - assert all(coello.bounds.lower == LB) - assert all(coello.bounds.upper == UB) - assert coello.bounds.snow == Snow - assert coello.bounds.maxbas is False - def test_read_gauge_table( self, coello_start_date: str, diff --git a/tests/rrm/catchment/test_run_narrowing.py b/tests/rrm/catchment/test_run_narrowing.py index 02c45549..eecda2e5 100644 --- a/tests/rrm/catchment/test_run_narrowing.py +++ b/tests/rrm/catchment/test_run_narrowing.py @@ -78,8 +78,12 @@ def test_a_finished_model_narrows(self, built: Catchment): run = DistributedRun.from_model(built) for field in ("period", "meteo", "flow_network", "parameters", "model_setup"): - assert getattr(run, field) is not None, f"{field} must be settled on the run" - assert run.meteo is built.meteo, "the run carries the model's own inputs, not copies" + assert getattr(run, field) is not None, ( + f"{field} must be settled on the run" + ) + assert run.meteo is built.meteo, ( + "the run carries the model's own inputs, not copies" + ) @pytest.mark.parametrize( "missing, expected", From 3d90669bf157bdf3af8b4ee925c9ca780868cf9a Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Sun, 6 Sep 2026 19:29:52 +0200 Subject: [PATCH 12/54] refactor(protocols): state the spatial-distribution contract, and finish calibration's types The calibration entry points typed `spatial_var_fun` as `Callable[..., Any]`, which was doubly wrong: a calibration never calls it, and what it actually does is read four members off it -- `Function`, `Par3d`, `no_parameters`, `no_elem`. So the annotation described a function while the code used an object, and there was nothing for mypy to check the four accesses against. `SpatialDistribution` states them. `Parameters` now provably satisfies it, which took two annotations there: `strategies` is typed `dict[int, Callable[..., Any]]` and `Function` declared rather than inferred, or mypy types the attribute from its first assignment and then rejects both the HRU reassignment below it and the protocol itself. With that and a set of narrowing accessors -- `_search_space`, `_objective`, `_gauged_results` -- `hapi.calibration` type-checks clean and comes off the suppression list, which leaves only `hapi.catchment`: the builder itself, where `X | None` is the honest description. Every module that *consumes* a catchment now narrows first. The accessors also replace a dozen reads that would have failed on `None` with errors naming the reader to call, and mypy caught two real problems on the way: `Catchment` had stopped satisfying `CatchmentLike` because the previous commit moved `bounds` off it without updating the protocol, and `run_calibration` still carried an inline copy of the grid checks that `DistributedRun.from_model` owns. That second one needed care. Deleting the inline copy moved the failure from before the optimiser was built to its first trial, which a test correctly objected to -- a search that cannot complete should not start. So `_check_before_optimising` calls the same seam early rather than repeating its checks, and reports a grid mismatch ahead of a missing objective function, because the first is a data problem the caller can act on. Three more tests were passing on swallowed failures, in the same way the previous commit found two: they never read an objective function or any observed discharge, so every trial raised on `None` and the assertions were reading the stubbed optimiser rather than a run. Fixed by giving them the inputs a calibration needs. --- pyproject.toml | 16 +- src/hapi/calibration.py | 196 ++++++++++++------ src/hapi/protocols.py | 35 +++- src/hapi/rrm/parameters.py | 9 +- .../test_calibration_distributed.py | 27 +++ 5 files changed, 210 insertions(+), 73 deletions(-) diff --git a/pyproject.toml b/pyproject.toml index 9e153f1a..2786293f 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -187,17 +187,13 @@ exclude = [ [[tool.mypy.overrides]] module = [ "hapi.catchment", - "hapi.calibration", ] -# The builder side. `Catchment` is assembled by successive read_*() calls, so its inputs -# are X | None until the matching one has run -- which mypy cannot verify and which is -# honest rather than a defect. `Calibration` reads those same optional attributes off the -# model it holds, and its `SpatialVarFun` argument is typed `Callable` though the code uses -# `.Function` / `.Par3d` / `.no_parameters` on it; a Protocol for that would close most of -# the remaining 44. -# -# The run layer no longer needs the excuse: `hapi.runs` narrows a builder into a validated, -# non-optional run, so `run`, `wrapper` and `distrrm` all type-check clean. +# The builder itself, and the last module on this list. `Catchment` is assembled by +# successive read_*() calls, so its inputs are X | None until the matching one has run -- +# which mypy cannot verify and which is honest rather than a defect. Everything that +# *consumes* a catchment now narrows first and type-checks clean: `run`, `wrapper` and +# `distrrm` through `hapi.runs`, and `calibration` through its own accessors plus the +# `SpatialDistribution` protocol. disable_error_code = [ "union-attr", "attr-defined", diff --git a/src/hapi/calibration.py b/src/hapi/calibration.py index 44832bf2..e4be50ee 100644 --- a/src/hapi/calibration.py +++ b/src/hapi/calibration.py @@ -18,6 +18,9 @@ from hapi.catchment import Catchment from hapi.conceptual import ParameterBounds, ParameterSet +from hapi.inputs import MeteoInputs +from hapi.protocols import SpatialDistribution +from hapi.results import SimulationResults from hapi.runs import DistributedRun, LumpedRun from hapi.wrapper import Wrapper @@ -165,20 +168,23 @@ def _declare_the_parameter_variables( # One starting value per parameter. A shorter list used to index out of range # part-way through building the problem, naming neither argument and leaving # `opt_prob` half-populated. - seeded = initial_values is not None and len(initial_values) > 0 - if seeded and len(initial_values) != len(self.bounds): + bounds = self._search_space() + # Bound rather than re-tested: `initial_values is not None` twice does not carry the + # narrowing into the indexing below, and the empty list is the "not seeded" case. + seeds = list(initial_values) if initial_values else [] + if seeds and len(seeds) != len(bounds): raise ValueError( f"initial_values must hold one value per parameter; the bounds define " - f"{len(self.bounds)} and {len(initial_values)} were given" + f"{len(bounds)} and {len(seeds)} were given" ) - for i in range(len(self.bounds)): - seed = {"value": initial_values[i]} if seeded else {} + for i in range(len(bounds)): + seed = {"value": seeds[i]} if seeds else {} opt_prob.addVar( f"x{i}", type="c", - lower=self.bounds.lower[i], - upper=self.bounds.upper[i], + lower=bounds.lower[i], + upper=bounds.upper[i], **seed, ) @@ -205,6 +211,87 @@ def _parameter_set(self, values) -> ParameterSet: maxbas = bounds.maxbas if bounds is not None else False return ParameterSet(values, snow=snow, maxbas=maxbas) + def _check_before_optimising(self, **narrowing: Any) -> None: + """Fail before the optimiser is built rather than on its first trial. + + Calls the same seam the objective function calls -- so there is still one place the + checks live -- just earlier, because starting a search that cannot possibly complete + wastes however long the first trial takes to reach the mismatch. + + The parameter array is skipped when unread: a calibration derives it from the bounds, + so there may be nothing to narrow yet, and the first trial checks it then. + + Args: + **narrowing: Forwarded to :meth:`~hapi.runs.DistributedRun.from_model`. + + Raises: + ValueError: The objective function is unread, or the model's inputs disagree. + """ + # The model first: a grid that does not line up is a data problem, and reporting it + # ahead of a missing setup step is what a caller can act on. + if self.model.parameters is not None: + DistributedRun.from_model(self.model, **narrowing) + self._objective() + + def _search_space(self) -> ParameterBounds: + """Return the bounds, or say which reader supplies them. + + The three entry points and the variable declaration all need them, and all used to + index `self.bounds` straight -- so a caller who forgot got a `TypeError` on `None` + part-way through building the optimisation problem. + + Returns: + ParameterBounds: The search space. + + Raises: + ValueError: The bounds have not been read. + """ + if self.bounds is None: + raise ValueError( + "the search space has not been read; call read_parameters_bound before " + "starting a calibration" + ) + return self.bounds + + def _objective(self) -> tuple[Callable[..., Any], list]: + """Return the objective function and its extra arguments. + + Returns: + tuple[Callable, list]: The metric and the arguments forwarded to it. + + Raises: + ValueError: No objective function has been read. + """ + if self.objective_function is None: + raise ValueError( + "there is no objective function to calibrate against; call " + "read_objective_function first" + ) + return self.objective_function, self.OFArgs or [] + + def _gauged_results(self) -> tuple[SimulationResults, MeteoInputs, Any]: + """Return the finished run and the gauge table its hydrographs are read at. + + Returns: + tuple: The results, the drivers (which size the series), and the gauge table. + + Raises: + ValueError: The model has not been run, or the gauges have not been read. + """ + results = self.model.results + if results is None: + raise ValueError( + "there are no results to extract; the calibration runs the model itself, so " + "this means no trial has completed" + ) + if self.model.meteo is None: + raise ValueError("the model has no drivers; assign model.meteo first") + if self.model.GaugesTable is None: + raise ValueError( + "the gauge table has not been read; call model.read_gauge_table first" + ) + return results, self.model.meteo, self.model.GaugesTable + def read_objective_function( self, objective_function: Callable[..., Any], args: list | None ): @@ -262,7 +349,13 @@ def extract_discharge( ValueError: The results came from MAXBAS routing, whose per-cell values are contributions rather than discharges. """ - if not self.model.results.outlet_shortcut_valid: + results, meteo, gauges = self._gauged_results() + if results.q_total is None: + raise ValueError( + "the results carry no routed discharge; the run did not complete" + ) + q_total = results.q_total + if not results.outlet_shortcut_valid: raise ValueError( "this catchment was run with triangular (MAXBAS) routing, which sends " "every cell straight to the outlet: a single cell of q_total is that cell's " @@ -271,24 +364,18 @@ def extract_discharge( "calibrated against the wrong signal." ) - self.Qsim = np.zeros((self.model.meteo.time_steps, len(self.model.GaugesTable))) + self.Qsim = np.zeros((meteo.time_steps, len(gauges))) # error = 0 - for i in range(len(self.model.GaugesTable)): - Xind = int( - self.model.GaugesTable.loc[self.model.GaugesTable.index[i], "cell_row"] - ) - Yind = int( - self.model.GaugesTable.loc[self.model.GaugesTable.index[i], "cell_col"] - ) + for i in range(len(gauges)): + Xind = int(gauges.loc[gauges.index[i], "cell_row"]) + Yind = int(gauges.loc[gauges.index[i], "cell_col"]) # gaugeid = self.model.GaugesTable.loc[self.model.GaugesTable.index[i],"id"] # Quz = self.model.results.quz_routed[Xind,Yind,:-1] # Qlz = self.model.results.qlz_translated[Xind,Yind,:-1] # self.Qsim[:,i] = Quz + Qlz - Qsim = np.reshape( - self.model.results.q_total[Xind, Yind, :-1], self.model.meteo.time_steps - ) + Qsim = np.reshape(q_total[Xind, Yind, :-1], meteo.time_steps) if factor is not None: self.Qsim[:, i] = Qsim * factor[i] @@ -302,7 +389,7 @@ def extract_discharge( def run_calibration( self, - spatial_var_fun: Callable[..., Any], + spatial_var_fun: SpatialDistribution, optimization_args: list, print_error: int | None = None, ): @@ -326,10 +413,9 @@ def run_calibration( gauge metadata. Args: - spatial_var_fun: Spatial variable function object with a - `Function` method that distributes parameters and a - `Par3d` attribute holding the 3D parameter array, plus - `no_parameters` and `no_elem` attributes. + spatial_var_fun: The spatial-distribution object that maps the optimiser's flat + vector onto the model's grid. See :class:`~hapi.protocols.SpatialDistribution` + for the four members read off it. optimization_args: A list of three elements: - `optimization_args[0]` (dict): Harmony Search API objective arguments (e.g., HMS, HMCR, PAR). @@ -351,22 +437,9 @@ def run_calibration( TypeError: If either bundle of optimization arguments is not a dict. """ - # input dimensions - # [rows,cols] = self.FlowAcc.ReadAsArray().shape - [fd_rows, fd_cols] = self.model.flow_network.flow_dir_arr.shape - if ( - fd_rows != self.model.flow_network.rows - or fd_cols != self.model.flow_network.cols - ): - raise ValueError(ROWS_MISMATCH_ERROR) - - # The three cubes already agree with each other (checked when MeteoInputs was - # built); this is the other half -- that they cover the model's grid. - self.model.meteo.validate_against( - self.model.flow_network.rows, - self.model.flow_network.cols, - self.model.period.date_index, - ) + # No dimension checks here: `DistributedRun.from_model` in the objective below is the + # single seam that makes them, and it runs outside the try, so the first trial surfaces + # a mismatch. Repeating them here is the drift the seam exists to stop. # basic inputs # check if all inputs are included @@ -382,6 +455,7 @@ def run_calibration( # check optimization arguement _check_optimization_args(api_obj_args, api_solve_args) + self._check_before_optimising() print("Calibration starts") ### calculate the objective function @@ -394,11 +468,12 @@ def opt_fun(par): self.model.parameters = self._parameter_set(spatial_var_fun.Par3d) run = DistributedRun.from_model(self.model) + objective, of_args = self._objective() try: self.model.results = Wrapper.run_muskingum(run) # calculate performance of the model try: - error = self.objective_function( + error = objective( self.model.QGauges, *[self.model.GaugesTable] ) # self.model.results.qout, self.model.results.quz_routed, self.model.results.qlz_translated, f = list(range(9, len(par), spatial_var_fun.no_parameters)) @@ -464,7 +539,7 @@ def opt_fun(par): def calibrate_maxbas( self, - spatial_var_fun: Callable[..., Any], + spatial_var_fun: SpatialDistribution, optimization_args: list, print_error: int | None = None, ): @@ -485,9 +560,8 @@ def calibrate_maxbas( gauge metadata. Args: - spatial_var_fun: Spatial variable function object with a - `Function` method that distributes parameters and a - `Par3d` attribute holding the 3D parameter array. + spatial_var_fun: The spatial-distribution object. See + :class:`~hapi.protocols.SpatialDistribution`. optimization_args: A list of three elements: - `optimization_args[0]` (dict): Harmony Search API objective arguments (e.g., HMS, HMCR, PAR). @@ -514,13 +588,7 @@ def calibrate_maxbas( # [fd_rows,fd_cols] = self.flow_dir_arr.shape # assert fd_rows == self.rows and fd_cols == self.cols, ROWS_MISMATCH_ERROR - # The three cubes already agree with each other (checked when MeteoInputs was - # built); this is the other half -- that they cover the model's grid. - self.model.meteo.validate_against( - self.model.flow_network.rows, - self.model.flow_network.cols, - self.model.period.date_index, - ) + # See run_calibration: the checks live in `DistributedRun.from_model`. # basic inputs # check if all inputs are included @@ -536,6 +604,7 @@ def calibrate_maxbas( # check optimization arguement _check_optimization_args(api_obj_args, api_solve_args) + self._check_before_optimising(needs_flow_direction=False) print("Calibration starts") # calculate the objective function @@ -546,11 +615,12 @@ def opt_fun(par): self.model.parameters = self._parameter_set(spatial_var_fun.Par3d) run = DistributedRun.from_model(self.model, needs_flow_direction=False) + objective, of_args = self._objective() try: self.model.results = Wrapper.run_maxbas(run) # calculate performance of the model try: - error = self.objective_function( + error = objective( self.model.QGauges, self.model.results.qout, *[self.model.GaugesTable], @@ -677,6 +747,8 @@ def calibrate_lumped( # check optimization arguement _check_optimization_args(api_obj_args, api_solve_args) + # A lumped run has no grid to check, so only the objective is verified up front. + self._objective() print("Calibration starts") ### calculate the objective function @@ -685,15 +757,23 @@ def opt_fun(par): self.model.parameters = self._parameter_set(par) run = LumpedRun.from_model(self.model) + objective, of_args = self._objective() + observed = self.model.QGauges + if observed is None: + raise ValueError( + "there is no observed discharge to score against; call " + "model.read_discharge_gauges first" + ) try: - self.model.results = Wrapper.run_lumped(run, route, routing_fn) - self.Qsim = self.model.results.q_total + run_results = Wrapper.run_lumped(run, route, routing_fn) + self.model.results = run_results + self.Qsim = run_results.q_total # calculate performance of the model try: - error = self.objective_function( - self.model.QGauges[self.model.QGauges.columns[-1]], + error = objective( + observed[observed.columns[-1]], self.Qsim, - *self.OFArgs, + *of_args, ) g = [ 2 * par[-2] * par[-1] / self.model.period.dt, diff --git a/src/hapi/protocols.py b/src/hapi/protocols.py index 0e7088cb..fa75ff7f 100644 --- a/src/hapi/protocols.py +++ b/src/hapi/protocols.py @@ -21,11 +21,12 @@ from __future__ import annotations +from collections.abc import Callable from typing import Any, Protocol import numpy as np -from hapi.conceptual import ConceptualModelSetup, ParameterBounds, ParameterSet +from hapi.conceptual import ConceptualModelSetup, ParameterSet from hapi.inputs import FlowNetwork, MeteoInputs, RiverGeometry from hapi.period import SimulationPeriod from hapi.results import SimulationResults @@ -46,7 +47,6 @@ class CatchmentLike(Protocol): data: The lumped driver record, once `read_lumped_inputs` has run. river_geometry: The hydraulic rasters, once `read_river_geometry` has run. flow_path_length_arr: The flow-path-length raster, once read. - bounds: The calibration search space, once `read_parameters_bound` has run. routing_method: Which routing the parameter set was calibrated for. results: Where a run's output lands. `None` before the first run. """ @@ -59,7 +59,6 @@ class CatchmentLike(Protocol): data: np.ndarray | None river_geometry: RiverGeometry | None flow_path_length_arr: np.ndarray | None - bounds: ParameterBounds | None routing_method: str results: SimulationResults | None @@ -72,3 +71,33 @@ class SupportsQsim(CatchmentLike, Protocol): """ Qsim: Any + + +class SpatialDistribution(Protocol): + """What a calibration needs of the thing that maps a flat vector onto the model's grid. + + The optimiser searches over a flat vector; the model runs on a `(rows, cols, n)` array. A + spatial-distribution object is what converts one into the other, and + :class:`hapi.rrm.parameters.Parameters` is the implementation that ships here. + + The calibration entry points used to type this argument `Callable[..., Any]`, which was + doubly wrong: a calibration never calls it, and what it actually does is read four members + off it. So the annotation described a function while the code used an object, and mypy had + nothing to check the four accesses against. + + Attributes: + Function: The distribution strategy, chosen by `Parameters.__init__` from the `function` + argument -- an attribute holding a callable rather than a method, which is why it is + declared as one. A calibration calls it with the trial vector alone; callers outside + may pass `kub` / `klb` too, hence the open signature. + Par3d: The `(rows, cols, no_parameters)` array `Function` fills in. Read straight after + each call, so the two are a pair: calling `Function` is what makes this current. + no_parameters: Parameters per cell. Strides the Muskingum K/X pairs out of the trial + vector when the constraints are built. + no_elem: Cells inside the domain. Sizes the two inequality constraints per cell. + """ + + Function: Callable[..., Any] + Par3d: np.ndarray + no_parameters: int + no_elem: int diff --git a/src/hapi/rrm/parameters.py b/src/hapi/rrm/parameters.py index 7739eb8c..060db45c 100644 --- a/src/hapi/rrm/parameters.py +++ b/src/hapi/rrm/parameters.py @@ -11,6 +11,8 @@ import datetime as dt import os import warnings +from collections.abc import Callable +from typing import Any import numpy as np from pyramids.dataset import Dataset @@ -176,7 +178,7 @@ def __init__( # Reject an unrecognised selector here rather than leaving `Function` unbound: # it is invoked on every calibration iteration, so a silent miss surfaces far # from the mistake as a bare AttributeError. - strategies = { + strategies: dict[int, Callable[..., Any]] = { 1: self.par3d_lumped, 2: self.par3d, 3: self.par2d_lumped_k1_lake, @@ -274,7 +276,10 @@ def __init__( shape=(self.no_parameters, self.no_elem), dtype=np.float32 ) - self.Function = strategies[function] + # Annotated, not inferred: mypy would otherwise type this from the one assignment and + # refuse the HRU reassignment below, and refuse it as a + # `hapi.protocols.SpatialDistribution` -- which is the contract a calibration reads. + self.Function: Callable[..., Any] = strategies[function] # to overwrite any choice user choose if the is HRUs if self.HRUs == 1: self.Function = self.hydrologic_response_units diff --git a/tests/rrm/calibration/test_calibration_distributed.py b/tests/rrm/calibration/test_calibration_distributed.py index a069431e..6b64f101 100644 --- a/tests/rrm/calibration/test_calibration_distributed.py +++ b/tests/rrm/calibration/test_calibration_distributed.py @@ -407,6 +407,8 @@ def test_stores_the_optimizer_result_on_the_instance( self, coello_rrm_date: list, lumped_meteo_data_path: str, + lumped_gauges_path: str, + coello_gauges_date_fmt: str, stub_optimizer: dict, ): """Test that the lumped entry point writes back `parameters` and `OFvalue`. @@ -421,6 +423,13 @@ def test_stores_the_optimizer_result_on_the_instance( # Without this the trials failed and the bare `except` scored them all `nan`; the stub # returns its canned result regardless, so the test passed over a model that never ran. coello.model.read_lumped_model(HBVLumped, 1530, [0, 10, 10, 10, 0]) + # Likewise: with no objective function, and nothing observed to score against, every + # trial raised on `None` and the bare `except` scored it `nan` -- so the assertions + # were reading the stub, not a run. + coello.model.read_discharge_gauges( + lumped_gauges_path, fmt=coello_gauges_date_fmt + ) + coello.read_objective_function(metrics.rmse, []) coello.bounds = ParameterBounds(np.zeros(12), np.ones(12)) basic_inputs = dict( Route=0, RoutingFn=Routing.triangular_routing_1, InitialValues=[] @@ -445,6 +454,8 @@ def test_initial_values_are_seeded_into_the_problem( self, coello_rrm_date: list, lumped_meteo_data_path: str, + lumped_gauges_path: str, + coello_gauges_date_fmt: str, stub_optimizer: dict, ): """Test that `InitialValues` reaches the optimisation problem as a starting point. @@ -459,6 +470,13 @@ def test_initial_values_are_seeded_into_the_problem( # Without this the trials failed and the bare `except` scored them all `nan`; the stub # returns its canned result regardless, so the test passed over a model that never ran. coello.model.read_lumped_model(HBVLumped, 1530, [0, 10, 10, 10, 0]) + # Likewise: with no objective function, and nothing observed to score against, every + # trial raised on `None` and the bare `except` scored it `nan` -- so the assertions + # were reading the stub, not a run. + coello.model.read_discharge_gauges( + lumped_gauges_path, fmt=coello_gauges_date_fmt + ) + coello.read_objective_function(metrics.rmse, []) coello.bounds = ParameterBounds(np.zeros(12), np.ones(12)) basic_inputs = dict( Route=0, @@ -476,6 +494,8 @@ def test_a_mismatched_initial_values_length_is_refused( self, coello_rrm_date: list, lumped_meteo_data_path: str, + lumped_gauges_path: str, + coello_gauges_date_fmt: str, stub_optimizer: dict, ): """Test that `InitialValues` shorter than the bounds is rejected, not indexed out of range. @@ -496,6 +516,13 @@ def test_a_mismatched_initial_values_length_is_refused( # Without this the trials failed and the bare `except` scored them all `nan`; the stub # returns its canned result regardless, so the test passed over a model that never ran. coello.model.read_lumped_model(HBVLumped, 1530, [0, 10, 10, 10, 0]) + # Likewise: with no objective function, and nothing observed to score against, every + # trial raised on `None` and the bare `except` scored it `nan` -- so the assertions + # were reading the stub, not a run. + coello.model.read_discharge_gauges( + lumped_gauges_path, fmt=coello_gauges_date_fmt + ) + coello.read_objective_function(metrics.rmse, []) coello.bounds = ParameterBounds(np.zeros(12), np.ones(12)) basic_inputs = dict( Route=0, From d22df14c889388840b4e6c7ae1e16370ab2664ac Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Sun, 6 Sep 2026 20:34:50 +0200 Subject: [PATCH 13/54] perf(routing): index the cells by accumulation value instead of rescanning the grid Muskingum routing must visit cells upstream-first, and flow accumulation gives that order. `route_muskingum` implemented "process the cells at level j" as "walk every cell in the grid and test whether it is at level j" -- once per level. Since the number of distinct accumulation levels grows with the domain, that made the pass O(n_acc x rows x cols), effectively quadratic, to visit each cell once. `FlowNetwork.cells_by_acc_val` groups the in-domain cells by accumulation value, cached beside `acc_val` and dropped with it when the array is replaced. Both loops now iterate the buckets, so the pass is O(no_elem). Measured rather than argued: - bit-identical output. `quz_routed`, `qlz_translated` and `q_total` all compare `array_equal` against the previous loop, diffed by loading the engine as it stood at the parent commit and running both over the Coello example. - Coello: 4,004 cell tests -> 89, and the pass 9.86 ms -> 2.18 ms (4.5x). - the saving grows with the grid, because what was removed is the quadratic term: 169x at 13x13, 2,500x at 50x50, 62,500x at 250x250, where 2.5 billion tests become 40,581. Order within a level stays row-major, matching the `x`-outer/`y`-inner scan it replaces -- that is what makes it bit-identical, since the routing accumulates and a different order inside a level could change the result. Four tests pin it: the buckets partition the domain exactly, the order is row-major, the visit count is linear, and the cache is invalidated with the array it derives from. The `# TODO parallelize` above the loop is now straightforward to act on: each bucket is an explicit, independent work list rather than a predicate spread over a grid scan. One thing found and deliberately left: the loop compares a float raster against `acc_val`, whose entries are truncated ints, so a fractional accumulation value (1.2 against the code 1) matches nothing and that cell is never routed. Both shipped datasets are integral, so it is latent. The index keys on the raw value so this behaves exactly as before -- fixing it would change results, which is a separate decision from this one. --- src/hapi/inputs.py | 46 +++++++++++- src/hapi/rrm/distrrm.py | 96 +++++++++++------------- tests/rrm/catchment/test_flow_network.py | 87 +++++++++++++++++++++ 3 files changed, 174 insertions(+), 55 deletions(-) diff --git a/src/hapi/inputs.py b/src/hapi/inputs.py index 7898073c..50a3742e 100644 --- a/src/hapi/inputs.py +++ b/src/hapi/inputs.py @@ -487,7 +487,7 @@ def no_elem(self) -> int: return int(np.count_nonzero(~np.isnan(self.flow_acc_arr))) def __setattr__(self, name: str, value: object) -> None: - """Drop the cached `acc_val` when the array it is derived from is replaced. + """Drop the caches derived from the accumulation array when it is replaced. Args: name: Attribute being set. @@ -495,6 +495,7 @@ def __setattr__(self, name: str, value: object) -> None: """ if name == "flow_acc_arr": self.__dict__.pop("acc_val", None) + self.__dict__.pop("cells_by_acc_val", None) object.__setattr__(self, name, value) @cached_property @@ -525,6 +526,49 @@ def acc_val(self) -> list[int]: values: list[int] = np.unique(_to_int_codes(self.flow_acc_arr)).tolist() return values + @cached_property + def cells_by_acc_val(self) -> dict[float, list[tuple[int, int]]]: + """dict: In-domain cell indices grouped by their accumulation value, row-major. + + The routing has to visit cells upstream-first, and accumulation gives that order. Asking + "which cells are at level j" used to be answered by walking the whole grid and testing + every cell against j -- once per level. Since the number of distinct levels grows with + the domain, that made the routing pass O(n_acc x rows x cols), effectively quadratic, to + visit each cell once. On the 13x14 Coello grid it was 4,004 visits for 89 cells; at + 250x250 it projects to about 1.9 billion. + + Building the answer once makes the same pass O(no_elem). Cached for the same reason + :attr:`acc_val` is, and dropped with it when `flow_acc_arr` is replaced. + + Keys are the raw accumulation values, not truncated ones, so a lookup by an + :attr:`acc_val` entry matches exactly the cells the old `==` test matched -- including + the case it missed. A fractional accumulation raster (1.2, say) yields the code `1` in + `acc_val` and so selects nothing, here as before; that is a separate question from this + one, and both shipped datasets are integral. + + Order within a level is row-major, matching the `x`-outer/`y`-inner scan it replaces, so + the routing visits cells in exactly the order it did. + + Examples: + >>> import numpy as np + >>> from hapi.inputs import FlowNetwork + >>> acc = np.array([[0.0, 1.0], [1.0, np.nan]]) + >>> network = FlowNetwork( + ... acc, no_data_value=-9999.0, cell_size=4000.0, px_area=16.0 + ... ) + >>> network.cells_by_acc_val[0.0] + [(0, 0)] + >>> network.cells_by_acc_val[1.0] + [(0, 1), (1, 0)] + + """ + grouped: dict[float, list[tuple[int, int]]] = {} + rows, cols = np.nonzero(~np.isnan(self.flow_acc_arr)) + for x, y in zip(rows, cols): + key = float(self.flow_acc_arr[x, y]) + grouped.setdefault(key, []).append((int(x), int(y))) + return grouped + @property def outlet(self) -> tuple: """tuple: Index of the most-accumulated cell, as `np.where` returns it.""" diff --git a/src/hapi/rrm/distrrm.py b/src/hapi/rrm/distrrm.py index 416d5587..311aae37 100644 --- a/src/hapi/rrm/distrrm.py +++ b/src/hapi/rrm/distrrm.py @@ -127,66 +127,54 @@ def route_muskingum(run: DistributedRun, results: SimulationResults) -> None: # in the catchment results.qlz_translated = np.zeros_like(results.quz) + + # Cells grouped by accumulation value, built once. Both loops below used to answer + # "which cells are at this level?" by walking the whole grid and testing every cell -- + # the second one doing it once per level, which made this pass O(n_acc x rows x cols) + # to visit each cell once. Row-major within a level, so the visit order is unchanged. + cells_by_acc_val = run.flow_network.cells_by_acc_val + # for all cells with 0 flow acc put the quz - for x in range(run.flow_network.rows): # no of rows - for y in range(run.flow_network.cols): # no of columns - if ( - not np.isnan(run.flow_network.flow_acc_arr[x, y]) - and run.flow_network.flow_acc_arr[x, y] == 0 - ): - results.quz_routed[x, y, :] = results.quz[x, y, :] - results.qlz_translated[x, y, :] = results.qlz[x, y, :] + for x, y in cells_by_acc_val.get(0, ()): + results.quz_routed[x, y, :] = results.quz[x, y, :] + results.qlz_translated[x, y, :] = results.qlz[x, y, :] # remaining cells - # Read once: this is the routing inner loop, and `acc_val` scans the whole grid. acc_val = run.flow_network.acc_val - for j in range(1, len(acc_val)): + for level in acc_val[1:]: # TODO parallelize # all cells with the same acc_val can run at the same time - for x in range(run.flow_network.rows): # no of rows - for y in range(run.flow_network.cols): # no of columns - # check from total flow accumulation - if ( - not np.isnan(run.flow_network.flow_acc_arr[x, y]) - and run.flow_network.flow_acc_arr[x, y] == acc_val[j] - ): - if river_depth is not None and river_depth[x, y] > 0: - # A river cell a 1D hydraulic model will route instead. The - # caller says so explicitly; this used to be inferred from - # `routing_method != "Muskingum"`, which meant any catchment - # built with a non-Muskingum method dereferenced - # `bankfull_depth` -- None outside the flood model -- and - # crashed here. - continue - else: - # for UZ - q_uzi = np.zeros(run.meteo.simulation_steps) - # for lz - qlzi = np.zeros(run.meteo.simulation_steps) - # iterate to route uz and translate lz - for i in range( - len(run.routing_table[str(x) + "," + str(y)]) - ): - # bring the indexes of the us cell - x_ind = run.routing_table[str(x) + "," + str(y)][i][0] - y_ind = run.routing_table[str(x) + "," + str(y)][i][1] - # sum the Q of the US cells (already routed for its cell) - # route first with there own k & xthen sum - q_uzi = q_uzi + routing.muskingum_v( - results.quz_routed[x_ind, y_ind, :], - results.quz_routed[x_ind, y_ind, 0], - run.parameter_cube[x_ind, y_ind, 10], - run.parameter_cube[x_ind, y_ind, 11], - run.period.dt, - ) - - qlzi = qlzi + results.qlz_translated[x_ind, y_ind, :] - - # add the routed upstream flows to the current Quz in the cell - results.quz_routed[x, y, :] = results.quz[x, y, :] + q_uzi - results.qlz_translated[x, y, :] = ( - results.qlz[x, y, :] + qlzi - ) + for x, y in cells_by_acc_val.get(level, ()): + if river_depth is not None and river_depth[x, y] > 0: + # A river cell a 1D hydraulic model will route instead. The caller says + # so explicitly; this used to be inferred from + # `routing_method != "Muskingum"`, which meant any catchment built with a + # non-Muskingum method dereferenced `bankfull_depth` -- None outside the + # flood model -- and crashed here. + continue + cell = f"{x},{y}" + upstream = run.routing_table[cell] + # for UZ + q_uzi = np.zeros(run.meteo.simulation_steps) + # for lz + qlzi = np.zeros(run.meteo.simulation_steps) + # iterate to route uz and translate lz + for x_ind, y_ind in upstream: + # sum the Q of the US cells (already routed for its cell) + # route first with there own k & xthen sum + q_uzi = q_uzi + routing.muskingum_v( + results.quz_routed[x_ind, y_ind, :], + results.quz_routed[x_ind, y_ind, 0], + run.parameter_cube[x_ind, y_ind, 10], + run.parameter_cube[x_ind, y_ind, 11], + run.period.dt, + ) + + qlzi = qlzi + results.qlz_translated[x_ind, y_ind, :] + + # add the routed upstream flows to the current Quz in the cell + results.quz_routed[x, y, :] = results.quz[x, y, :] + q_uzi + results.qlz_translated[x, y, :] = results.qlz[x, y, :] + qlzi results.q_total = results.qlz_translated + results.quz_routed # Muskingum accumulates downstream, so a cell of `q_total` is the discharge at that # cell and the outlet-cell shortcut in `extract_discharge` is valid. diff --git a/tests/rrm/catchment/test_flow_network.py b/tests/rrm/catchment/test_flow_network.py index bbba5859..4ff764fe 100644 --- a/tests/rrm/catchment/test_flow_network.py +++ b/tests/rrm/catchment/test_flow_network.py @@ -352,3 +352,90 @@ def test_the_cached_value_still_matches_the_uncached_computation( assert network.acc_val == expected, ( f"Expected {expected}, got {network.acc_val}" ) + + +class TestCellsByAccVal: + """The bucketed index that makes the routing pass linear in the domain.""" + + def test_it_groups_every_in_domain_cell_exactly_once(self): + """Test that the buckets partition the domain -- no cell missed, none duplicated. + + Test scenario: + The routing visits cells through these buckets, so a missing cell is a cell that + never gets routed and a duplicated one is routed twice. The count is the invariant + that makes the replacement safe. + """ + acc = np.array([[0.0, 1.0, 1.0], [2.0, np.nan, 2.0], [3.0, 3.0, np.nan]]) + network = FlowNetwork( + acc, no_data_value=-9999.0, cell_size=4000.0, px_area=16.0 + ) + + grouped = network.cells_by_acc_val + total = sum(len(cells) for cells in grouped.values()) + + assert total == network.no_elem, ( + f"the buckets must hold every domain cell once; {total} against " + f"{network.no_elem} in the domain" + ) + flat = [cell for cells in grouped.values() for cell in cells] + assert len(set(flat)) == len(flat), "no cell may appear in two buckets" + + def test_order_within_a_level_is_row_major(self): + """Test that cells come back in the order the replaced scan visited them. + + Test scenario: + The loop this replaces was `for x: for y:`, so within one accumulation level it + went row by row. Muskingum routing accumulates, so a different order inside a + level could change the result -- this is what makes the change bit-identical. + """ + acc = np.array([[5.0, 5.0], [5.0, 5.0]]) + network = FlowNetwork( + acc, no_data_value=-9999.0, cell_size=4000.0, px_area=16.0 + ) + + assert network.cells_by_acc_val[5.0] == [(0, 0), (0, 1), (1, 0), (1, 1)], ( + "cells within a level must be row-major, matching the x-outer/y-inner scan" + ) + + def test_it_visits_far_fewer_cells_than_a_per_level_grid_scan(self): + """Test that the index is what removes the quadratic term. + + Test scenario: + Answering "which cells are at level j" by scanning the grid costs + `n_acc x rows x cols`; the number of levels grows with the domain, so that is + effectively quadratic. This pins the linear count, so a change back to a scan + shows up here rather than as an unexplained slowdown on a real catchment. + """ + side = 12 + acc = np.arange(side * side, dtype=float).reshape(side, side) + network = FlowNetwork( + acc, no_data_value=-9999.0, cell_size=4000.0, px_area=16.0 + ) + + bucketed = sum(len(cells) for cells in network.cells_by_acc_val.values()) + scanned = len(network.acc_val) * network.rows * network.cols + + assert bucketed == network.no_elem, "one visit per domain cell" + assert scanned // bucketed >= side * side // 2, ( + f"the scan this replaces cost {scanned} tests against {bucketed} visits; if that " + "ratio has collapsed the index is no longer doing its job" + ) + + def test_replacing_the_accumulation_array_drops_the_index(self): + """Test that the cache is invalidated with `acc_val`, not left describing the old grid. + + Test scenario: + `acc_val` is already dropped on replacement. An index that survived would send the + routing to cell coordinates from a different grid, which is worse than recomputing. + """ + acc = np.array([[0.0, 1.0], [2.0, np.nan]]) + network = FlowNetwork( + acc, no_data_value=-9999.0, cell_size=4000.0, px_area=16.0 + ) + assert len(network.cells_by_acc_val) == 3, "primed from the first array" + + network.flow_acc_arr = np.array([[7.0, 7.0], [7.0, 7.0]]) + + assert network.cells_by_acc_val == {7.0: [(0, 0), (0, 1), (1, 0), (1, 1)]}, ( + "the index must be rebuilt from the replacement array" + ) From bbebd1c7082281d2ffb602d32cf04dbc1ccd9257 Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Sun, 6 Sep 2026 21:02:57 +0200 Subject: [PATCH 14/54] fix(routing): route cells whose accumulation value is not a whole number `FlowNetwork.acc_val` truncates to integer codes -- `np.unique(_to_int_codes(...))` -- but the routing selected cells by matching those codes against the *raw* accumulation value. So a raster holding 1.2 produced the code 1, `1.2 == 1` was False, and that cell was never routed. Worse than one cell reading zero: the routing sums `quz_routed` from each cell's upstream neighbours, so an unrouted cell contributed nothing to anything below it, and its whole tributary vanished from the hydrograph all the way to the outlet. Nothing raised. Demonstrated on a 2x2 domain holding [0.0, 1.2, 2.5, 3.0]: the loop selected 2 of the 4 domain cells and dropped (0, 1) and (1, 0). Not a design decision. `_to_int_codes` replaced a per-cell `set(int(...))` that truncated on *both* sides of the comparison; somewhere the lookup kept the truncation and the match did not. `acc_val` still documents the original intent -- "1.2 and 1.8 are one code" -- which only makes sense if such cells are then routed together at that code. They were not; they were skipped. So the docstring described the half that no longer happened. `cells_by_acc_val` now keys on the truncated code, restoring that contract: values sharing a code share a bucket and route together. Integral rasters are unaffected, re-verified bit-identical against the engine as it stood before the bucketing change -- and both shipped datasets are integral, so nothing that ships changes behaviour. What changes is a weighted accumulation raster, or one some tool exported as float, which previously lost cells in silence. --- src/hapi/inputs.py | 58 ++++++++++++++++-------- tests/rrm/catchment/test_flow_network.py | 42 +++++++++++++++++ 2 files changed, 81 insertions(+), 19 deletions(-) diff --git a/src/hapi/inputs.py b/src/hapi/inputs.py index 50a3742e..7fe75896 100644 --- a/src/hapi/inputs.py +++ b/src/hapi/inputs.py @@ -527,8 +527,8 @@ def acc_val(self) -> list[int]: return values @cached_property - def cells_by_acc_val(self) -> dict[float, list[tuple[int, int]]]: - """dict: In-domain cell indices grouped by their accumulation value, row-major. + def cells_by_acc_val(self) -> dict[int, list[tuple[int, int]]]: + """dict: In-domain cell indices grouped by their accumulation code, row-major. The routing has to visit cells upstream-first, and accumulation gives that order. Asking "which cells are at level j" used to be answered by walking the whole grid and testing @@ -540,32 +540,52 @@ def cells_by_acc_val(self) -> dict[float, list[tuple[int, int]]]: Building the answer once makes the same pass O(no_elem). Cached for the same reason :attr:`acc_val` is, and dropped with it when `flow_acc_arr` is replaced. - Keys are the raw accumulation values, not truncated ones, so a lookup by an - :attr:`acc_val` entry matches exactly the cells the old `==` test matched -- including - the case it missed. A fractional accumulation raster (1.2, say) yields the code `1` in - `acc_val` and so selects nothing, here as before; that is a separate question from this - one, and both shipped datasets are integral. + Keys are **truncated** to integers, the same codes :attr:`acc_val` holds, so every + in-domain cell is reachable through one of them. That matters for a fractional raster: + the grid scan this replaced compared the raw value against a truncated code, so `1.2` + never matched the code `1` and the cell was never routed at all -- and because the + routing sums `quz_routed` from upstream neighbours, that cell's whole contribution + vanished from every cell below it, silently. Truncating both sides is what + :attr:`acc_val` has always documented ("1.2 and 1.8 are one code"); only the comparison + had drifted. Order within a level is row-major, matching the `x`-outer/`y`-inner scan it replaces, so the routing visits cells in exactly the order it did. Examples: - >>> import numpy as np - >>> from hapi.inputs import FlowNetwork - >>> acc = np.array([[0.0, 1.0], [1.0, np.nan]]) - >>> network = FlowNetwork( - ... acc, no_data_value=-9999.0, cell_size=4000.0, px_area=16.0 - ... ) - >>> network.cells_by_acc_val[0.0] - [(0, 0)] - >>> network.cells_by_acc_val[1.0] - [(0, 1), (1, 0)] + - Integral accumulation, which is what a cell-count raster holds: + + >>> import numpy as np + >>> from hapi.inputs import FlowNetwork + >>> acc = np.array([[0.0, 1.0], [1.0, np.nan]]) + >>> network = FlowNetwork( + ... acc, no_data_value=-9999.0, cell_size=4000.0, px_area=16.0 + ... ) + >>> network.cells_by_acc_val[0] + [(0, 0)] + >>> network.cells_by_acc_val[1] + [(0, 1), (1, 0)] + + - Fractional values share the code they truncate to, so they route together + rather than being dropped: + + >>> import numpy as np + >>> from hapi.inputs import FlowNetwork + >>> acc = np.array([[1.2, 1.8], [3.0, np.nan]]) + >>> network = FlowNetwork( + ... acc, no_data_value=-9999.0, cell_size=4000.0, px_area=16.0 + ... ) + >>> network.acc_val + [1, 3] + >>> network.cells_by_acc_val[1] + [(0, 0), (0, 1)] """ - grouped: dict[float, list[tuple[int, int]]] = {} + grouped: dict[int, list[tuple[int, int]]] = {} rows, cols = np.nonzero(~np.isnan(self.flow_acc_arr)) for x, y in zip(rows, cols): - key = float(self.flow_acc_arr[x, y]) + # Truncated, matching `acc_val`: see the note above on the comparison that drifted. + key = int(self.flow_acc_arr[x, y]) grouped.setdefault(key, []).append((int(x), int(y))) return grouped diff --git a/tests/rrm/catchment/test_flow_network.py b/tests/rrm/catchment/test_flow_network.py index 4ff764fe..40f92d00 100644 --- a/tests/rrm/catchment/test_flow_network.py +++ b/tests/rrm/catchment/test_flow_network.py @@ -421,6 +421,48 @@ def test_it_visits_far_fewer_cells_than_a_per_level_grid_scan(self): "ratio has collapsed the index is no longer doing its job" ) + def test_a_fractional_raster_still_reaches_every_cell(self): + """Test that fractional accumulation values are routed, not silently dropped. + + Test scenario: + `acc_val` truncates, so a raster holding 1.2 yields the code 1. The grid scan this + index replaced compared the *raw* value against that code, so `1.2 == 1` was False + and the cell was never routed -- and since the routing sums `quz_routed` from + upstream neighbours, its entire contribution vanished from every cell downstream, + with nothing raised. Keying on the same truncated code is what `acc_val` has always + documented; only the comparison had drifted away from it. + """ + acc = np.array([[0.0, 1.2], [2.5, 3.0]]) + network = FlowNetwork( + acc, no_data_value=-9999.0, cell_size=4000.0, px_area=16.0 + ) + + selected = sum( + len(network.cells_by_acc_val.get(level, ())) for level in network.acc_val + ) + + assert selected == network.no_elem, ( + f"every domain cell must be reachable through an acc_val code; {selected} of " + f"{network.no_elem} were, so the rest would never be routed" + ) + + def test_values_sharing_a_code_are_grouped(self): + """Test that 1.2 and 1.8 land in one bucket, as `acc_val` documents. + + Test scenario: + `acc_val` collapses them to the single code 1, so the routing treats them as one + level. The bucket has to agree, or the level would select only some of its cells. + """ + acc = np.array([[1.2, 1.8], [3.0, np.nan]]) + network = FlowNetwork( + acc, no_data_value=-9999.0, cell_size=4000.0, px_area=16.0 + ) + + assert network.acc_val == [1, 3], "the two fractional values are one code" + assert network.cells_by_acc_val[1] == [(0, 0), (0, 1)], ( + "both cells at that code must be in its bucket" + ) + def test_replacing_the_accumulation_array_drops_the_index(self): """Test that the cache is invalidated with `acc_val`, not left describing the old grid. From 3dc3177eb55cd2e585dd74d0c33e27e32b92c2bb Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Sun, 6 Sep 2026 21:40:44 +0200 Subject: [PATCH 15/54] fix(inputs): guard FlowNetwork's grid invariant on replacement, and clear the plan Three remaining items from the coupling plan, all small. `FlowNetwork` promised its two rasters share a grid but only checked at construction: `__setattr__` dropped the derived caches and let a replacement of any shape through, so the accumulation and direction arrays could end up describing different catchments and a cell index would mean a different place in each. It now calls a `_check_replacement` mirroring the one `MeteoInputs` has always had. That makes the shape check in `DistributedRun.from_model` unreachable, so it is gone. It only ever existed to compensate for this gap -- the architecture review predicted it would become dead once the hole closed, and it has. The test that staged a mismatched raster to exercise it now asserts the refusal at the point of assignment, which is where the mistake is made. `read_discharge_gauges` reads `self.period.date_index` instead of re-deriving the daily/hourly branch, which was the last of four hand-written copies. `Wrapper` and `DistributedRRM` lose their do-nothing `__init__`. Both are namespaces of static methods and neither was ever instantiated. Verified: a mismatched replacement raises, a same-shape one is still allowed, and adding a direction raster to a network built without one still works -- the last mattering because MAXBAS builds networks with no direction raster at all. --- src/hapi/catchment.py | 7 +++-- src/hapi/inputs.py | 30 ++++++++++++++++++++++ src/hapi/rrm/distrrm.py | 4 --- src/hapi/runs.py | 11 +++----- src/hapi/wrapper.py | 4 --- tests/rrm/catchment/test_run_validation.py | 30 +++++++++++----------- 6 files changed, 51 insertions(+), 35 deletions(-) diff --git a/src/hapi/catchment.py b/src/hapi/catchment.py index 792ef437..ba78b423 100644 --- a/src/hapi/catchment.py +++ b/src/hapi/catchment.py @@ -928,10 +928,9 @@ def read_discharge_gauges( ValueError: If the gauge table has not been read yet (distributed mode). """ - if self.period.temporal_resolution.lower() == "daily": - ind = pd.date_range(self.period.start, self.period.end, freq="D") - else: - ind = pd.date_range(self.period.start, self.period.end, freq="h") + # The calendar belongs to the period, which derives it from the span and the + # resolution. This was the last of four hand-written copies of that branch. + ind = self.period.date_index if self.spatial_resolution.lower() == "distributed": self._read_one_discharge_file_per_gauge( diff --git a/src/hapi/inputs.py b/src/hapi/inputs.py index 7fe75896..59407399 100644 --- a/src/hapi/inputs.py +++ b/src/hapi/inputs.py @@ -493,11 +493,41 @@ def __setattr__(self, name: str, value: object) -> None: name: Attribute being set. value: New value. """ + if name in ("flow_acc_arr", "flow_dir_arr"): + self._check_replacement(name, value) if name == "flow_acc_arr": self.__dict__.pop("acc_val", None) self.__dict__.pop("cells_by_acc_val", None) object.__setattr__(self, name, value) + def _check_replacement(self, name: str, value: object) -> None: + """Reject a raster that would leave the two describing different grids. + + `__post_init__` alone cannot hold the class's promise that the two rasters share a + grid: the fields are plain mutable attributes, so replacing one afterwards silently + broke it, and a cell index then meant a different place in each. `MeteoInputs` has + guarded its cubes this way since it was written; this is the same guard, which the run + layer had been compensating for by re-checking the shape itself. + + Args: + name: Which raster is being replaced. + value: The replacement. + + Raises: + ValueError: The replacement is not a 2D array of the shape the other one has. + """ + other = "flow_dir_arr" if name == "flow_acc_arr" else "flow_acc_arr" + current = getattr(self, other, None) + if current is None or value is None: + return + expected = np.shape(current) + if not isinstance(value, np.ndarray) or value.shape != expected: + got = getattr(value, "shape", type(value).__name__) + raise ValueError( + f"{name} must stay {expected} to match {other}, got {got}; build a new " + "FlowNetwork to change the catchment's grid" + ) + @cached_property def acc_val(self) -> list[int]: """list[int]: The distinct accumulation values inside the domain, ascending. diff --git a/src/hapi/rrm/distrrm.py b/src/hapi/rrm/distrrm.py index 311aae37..a5b727f9 100644 --- a/src/hapi/rrm/distrrm.py +++ b/src/hapi/rrm/distrrm.py @@ -31,10 +31,6 @@ class DistributedRRM: catchment, so nothing here has to ask whether its inputs were checked. """ - def __init__(self): - """Distributed constructor.""" - pass - @staticmethod def run_lumped_model(run: DistributedRun) -> SimulationResults: """Run lumped rainfall-runoff model for every grid cell. diff --git a/src/hapi/runs.py b/src/hapi/runs.py index 95bc47c3..2a230d01 100644 --- a/src/hapi/runs.py +++ b/src/hapi/runs.py @@ -192,14 +192,9 @@ def from_model( "this run routes cell to cell and needs a flow-direction raster, but the " "flow network was built without one; pass it to FlowNetwork.from_rasters" ) - # `FlowNetwork.__post_init__` already checks this at construction, but - # `__setattr__` does not re-check on replacement (unlike `MeteoInputs`), so a - # raster swapped in afterwards can still disagree with the grid. - if flow_network.flow_dir_arr.shape != ( - flow_network.rows, - flow_network.cols, - ): - raise ValueError(GRID_MISMATCH_ERROR) + # No shape check here: `FlowNetwork` guards its own invariant at construction and + # on replacement now, so the two rasters cannot disagree. This compensated for the + # replacement gap, and became unreachable when that closed. if flow_network.FDT is None: raise ValueError( "cell-to-cell routing needs the flow-direction table; the flow network " diff --git a/src/hapi/wrapper.py b/src/hapi/wrapper.py index 695b318b..eeaa3f3f 100644 --- a/src/hapi/wrapper.py +++ b/src/hapi/wrapper.py @@ -69,10 +69,6 @@ class Wrapper: Lumped: Run a lumped conceptual model with optional routing. """ - def __init__(self): - """Initialize the Wrapper class.""" - pass - @staticmethod def run_muskingum(run: DistributedRun) -> SimulationResults: """Run the distributed rainfall-runoff model with spatial routing. diff --git a/tests/rrm/catchment/test_run_validation.py b/tests/rrm/catchment/test_run_validation.py index ac7f20aa..848b3926 100644 --- a/tests/rrm/catchment/test_run_validation.py +++ b/tests/rrm/catchment/test_run_validation.py @@ -243,27 +243,27 @@ def test_rejects_a_lake_record_missing_a_column( with pytest.raises(ValueError, match="three columns"): Run.run_distributed_with_lake(coello_loaded, lake) - def test_rejects_a_flow_direction_grid_of_the_wrong_shape( - self, coello_loaded: Catchment, spied_wrapper: dict + def test_a_flow_direction_grid_of_the_wrong_shape_cannot_be_installed( + self, coello_loaded: Catchment ): - """Test that the flow-direction grid is checked against the network's own shape. + """Test that the two rasters cannot be left describing different grids. Test scenario: - `FlowNetwork` sizes itself from the accumulation raster, so a flow-direction - array of a different shape means the two rasters do not describe the same - catchment and the routing table would be indexed out of range. + `FlowNetwork` sizes itself from the accumulation raster, so a flow-direction array + of a different shape means the two do not describe the same catchment and the + routing table would be indexed out of range. `__post_init__` checked that at + construction but `__setattr__` did not re-check on replacement, so this used to be + stageable and the run layer re-checked the shape to catch it. The guard now lives + where the mistake is made, which is why that run-layer check could go. """ - coello_loaded.flow_network.flow_dir_arr = ( - coello_loaded.flow_network.flow_dir_arr[:-1, :] - ) - lake = _LakeStub(coello_loaded.meteo.time_steps) - - with pytest.raises(ValueError, match="rows and columns"): - Run.run_distributed_with_lake(coello_loaded, lake) + network = coello_loaded.flow_network + with pytest.raises(ValueError, match="must stay"): + network.flow_dir_arr = network.flow_dir_arr[:-1, :] -class TestRunFW1WithLake: - """Tests for `Run.run_maxbas_with_lake`.""" + assert network.flow_dir_arr.shape == (network.rows, network.cols), ( + "the refused assignment must leave the network as it was" + ) def test_dispatches_once_the_lake_record_matches_the_simulation( self, coello_loaded: Catchment, spied_wrapper: dict From 73fd4a3ddce01240da0f1f499a1ab37291d91eae Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Sun, 6 Sep 2026 22:14:51 +0200 Subject: [PATCH 16/54] perf(results): stop allocating the state array on runs that never read it Half the memory a distributed run allocates was `state_variables`, a `(rows, cols, steps, 5)` array -- as much as `quz`, `qlz`, `quz_routed`, `qlz_translated` and `q_total` put together. Nothing reads it but `save_results` and `plot_distributed_results` options 4 to 8. The routing does not touch it, and neither does `extract_discharge` or any calibration. `DistributedRun.keep_state_variables` makes it optional. It defaults to True, so no existing caller changes behaviour or memory. `Calibration` passes False, which is where this matters: it runs the model once per trial vector, thousands of times per search, and never looks at the states. Measured on the Coello example: result arrays 78 KiB -> 39 KiB, exactly half, with all five discharge fields `array_equal` either way. Projected on a 500x500 grid over a decade of daily steps, 34 GiB -> 17 GiB. The state options now go through `_require_state_variables`, which names the switch instead of failing on `None` inside a slice several frames away. It is called at each option rather than bound once at the top of the method, so a discharge-only plot on such a run still works -- covered by a test, since binding it eagerly is the obvious way to write this and is wrong. Scope, plainly: this is the achievable half of ARC-HAPI-12, not the streaming the item originally proposed. Streaming the time axis is not reachable from here -- HBV's `simulate()` runs the whole series for one cell at a time, and the routing needs every cell's full series in upstream-to-downstream order, so no cell's history can be dropped mid-pass. Chunking time would mean restructuring the conceptual model to accept and return partial state, which is separate work. What is left of the ceiling is 5 cube-equivalents instead of 10. --- src/hapi/calibration.py | 10 ++- src/hapi/catchment.py | 52 ++++++++++--- src/hapi/results.py | 8 +- src/hapi/rrm/distrrm.py | 19 +++-- src/hapi/runs.py | 12 +++ tests/rrm/catchment/test_run_narrowing.py | 91 +++++++++++++++++++++++ 6 files changed, 170 insertions(+), 22 deletions(-) diff --git a/src/hapi/calibration.py b/src/hapi/calibration.py index e4be50ee..7c5f5fe5 100644 --- a/src/hapi/calibration.py +++ b/src/hapi/calibration.py @@ -466,7 +466,10 @@ def opt_fun(par): # search on over a model that never ran -- which is what the bare `except` did. spatial_var_fun.Function(par) self.model.parameters = self._parameter_set(spatial_var_fun.Par3d) - run = DistributedRun.from_model(self.model) + # The states are five times the size of every other result field and a + # calibration never reads them, so they are not allocated -- once per trial + # vector, that is half the peak memory of the whole search. + run = DistributedRun.from_model(self.model, keep_state_variables=False) objective, of_args = self._objective() try: @@ -613,7 +616,10 @@ def opt_fun(par): # vector or a grid mismatch surfaces instead of being scored `nan`. spatial_var_fun.Function(par) self.model.parameters = self._parameter_set(spatial_var_fun.Par3d) - run = DistributedRun.from_model(self.model, needs_flow_direction=False) + # See run_calibration: the states are not read, so they are not allocated. + run = DistributedRun.from_model( + self.model, needs_flow_direction=False, keep_state_variables=False + ) objective, of_args = self._objective() try: diff --git a/src/hapi/catchment.py b/src/hapi/catchment.py index ba78b423..f75c71b2 100644 --- a/src/hapi/catchment.py +++ b/src/hapi/catchment.py @@ -223,6 +223,30 @@ def _name_the_path(path) -> Iterator[None]: raise FileNotFoundError(f"{exc} (path: {path})") from exc +def _require_state_variables(results: SimulationResults) -> np.ndarray: + """Return the per-cell state array, or say why it is absent. + + It is `(rows, cols, time, 5)` -- as much memory as every other result field combined -- so a + run can be asked not to keep it. Only these plotting and saving options read it, so the + error belongs here, naming the switch rather than failing on `None` inside a slice. + + Args: + results: The run's results. + + Returns: + np.ndarray: The state array. + + Raises: + ValueError: The run was asked not to keep the states. + """ + if results.state_variables is None: + raise ValueError( + "this run did not keep the state variables, so no state option can be plotted or " + "saved; run it with keep_state_variables=True (the default) if you need them" + ) + return results.state_variables + + class Catchment: """Catchment for reading meteorological/spatial inputs and running the model. @@ -1364,19 +1388,19 @@ def plot_distributed_results( arr = self.results.qlz_translated[:, :, start_i:end_i] title = "Ground Water Flow" elif option == 4: - arr = self.results.state_variables[:, :, start_i:end_i, 0] + arr = _require_state_variables(self.results)[:, :, start_i:end_i, 0] title = "Snow Pack" elif option == 5: - arr = self.results.state_variables[:, :, start_i:end_i, 1] + arr = _require_state_variables(self.results)[:, :, start_i:end_i, 1] title = "Soil Moisture" elif option == 6: - arr = self.results.state_variables[:, :, start_i:end_i, 2] + arr = _require_state_variables(self.results)[:, :, start_i:end_i, 2] title = "Upper Zone" elif option == 7: - arr = self.results.state_variables[:, :, start_i:end_i, 3] + arr = _require_state_variables(self.results)[:, :, start_i:end_i, 3] title = "Lower Zone" elif option == 8: - arr = self.results.state_variables[:, :, start_i:end_i, 4] + arr = _require_state_variables(self.results)[:, :, start_i:end_i, 4] title = "Water Content" elif option == 9: arr = self.meteo.precipitation[:, :, start_i:end_i] @@ -1534,15 +1558,15 @@ def save_results( elif result == 3: arr = self.results.qlz_translated[:, :, start_i:end_i] elif result == 4: - arr = self.results.state_variables[:, :, start_i:end_i, 0] + arr = _require_state_variables(self.results)[:, :, start_i:end_i, 0] elif result == 5: - arr = self.results.state_variables[:, :, start_i:end_i, 1] + arr = _require_state_variables(self.results)[:, :, start_i:end_i, 1] elif result == 6: - arr = self.results.state_variables[:, :, start_i:end_i, 2] + arr = _require_state_variables(self.results)[:, :, start_i:end_i, 2] elif result == 7: - arr = self.results.state_variables[:, :, start_i:end_i, 3] + arr = _require_state_variables(self.results)[:, :, start_i:end_i, 3] elif result == 8: - arr = self.results.state_variables[:, :, start_i:end_i, 4] + arr = _require_state_variables(self.results)[:, :, start_i:end_i, 4] else: raise ValueError( f" The result parameter takes a value between 1 and 8, given: {result}" @@ -1571,13 +1595,17 @@ def save_results( data["Qlz"] = self.results.qlz[start_i:end_i] data.to_csv(path, index=False, float_format="%.3f") elif result == 4: - data[STATE_VARIABLES] = self.results.state_variables[start_i:end_i, :] + data[STATE_VARIABLES] = _require_state_variables(self.results)[ + start_i:end_i, : + ] data.to_csv(path, index=False, float_format="%.3f") elif result == 5: data["Qsim"] = self.Qsim[start_i:end_i] data["Quz"] = self.results.quz[start_i:end_i] data["Qlz"] = self.results.qlz[start_i:end_i] - data[STATE_VARIABLES] = self.results.state_variables[start_i:end_i, :] + data[STATE_VARIABLES] = _require_state_variables(self.results)[ + start_i:end_i, : + ] data.to_csv(path, index=False, float_format="%.3f") else: raise ValueError( diff --git a/src/hapi/results.py b/src/hapi/results.py index 3e6914e8..47b5c799 100644 --- a/src/hapi/results.py +++ b/src/hapi/results.py @@ -59,7 +59,11 @@ class SimulationResults: quz: `(rows, cols, time)` upper-zone discharge in m3/s. For a lumped run, a 1D series. qlz: `(rows, cols, time)` lower-zone discharge in m3/s. For a lumped run, a 1D series. state_variables: `(rows, cols, time, 5)` state array, the states being - `[sp, sm, uz, lz, wc]`. For a lumped run, `(time, 5)`. + `[sp, sm, uz, lz, wc]`. For a lumped run, `(time, 5)`. `None` when a distributed + run was asked not to keep them -- it is five times the size of every other field + put together and nothing but `save_results` and `plot_distributed_results` reads + it, so a run that will not look at it need not pay for it. See + :attr:`~hapi.runs.DistributedRun.keep_state_variables`. quz_routed: Upper-zone discharge after routing. `None` until a routing step runs. qlz_translated: Lower-zone discharge after translation. `None` until then. q_total: `quz_routed + qlz_translated`. Read it through @@ -104,7 +108,7 @@ class SimulationResults: routing: RoutingKind quz: np.ndarray qlz: np.ndarray - state_variables: np.ndarray + state_variables: np.ndarray | None quz_routed: np.ndarray | None = None qlz_translated: np.ndarray | None = None q_total: np.ndarray | None = None diff --git a/src/hapi/rrm/distrrm.py b/src/hapi/rrm/distrrm.py index a5b727f9..43857196 100644 --- a/src/hapi/rrm/distrrm.py +++ b/src/hapi/rrm/distrrm.py @@ -55,18 +55,19 @@ def run_lumped_model(run: DistributedRun) -> SimulationResults: routing=RoutingKind.UNROUTED, quz=np.zeros(grid, dtype=np.float32), qlz=np.zeros(grid, dtype=np.float32), - state_variables=np.zeros((*grid, 5), dtype=np.float32), + state_variables=( + np.zeros((*grid, 5), dtype=np.float32) + if run.keep_state_variables + else None + ), ) + states = results.state_variables for x in range(run.flow_network.rows): for y in range(run.flow_network.cols): # only for cells in the domain if not np.isnan(run.flow_network.flow_acc_arr[x, y]): - ( - results.quz[x, y, :], - results.qlz[x, y, :], - results.state_variables[x, y, :, :], - ) = run.model_setup.model.simulate( + quz_cell, qlz_cell, states_cell = run.model_setup.model.simulate( prec=run.meteo.precipitation[x, y, :], temp=run.meteo.temperature[x, y, :], et=run.meteo.evapotranspiration[x, y, :], @@ -76,6 +77,12 @@ def run_lumped_model(run: DistributedRun) -> SimulationResults: q_init=run.model_setup.q_init, snow=run.parameters.snow, ) + results.quz[x, y, :] = quz_cell + results.qlz[x, y, :] = qlz_cell + # Dropped rather than stored when the caller said it will not read them; + # the states are five times the size of everything else here. + if states is not None: + states[x, y, :, :] = states_cell area_coef = run.model_setup.area / run.flow_network.px_tot_area factor = run.flow_network.px_area * area_coef / run.period.conversion_factor diff --git a/src/hapi/runs.py b/src/hapi/runs.py index 2a230d01..61c09bee 100644 --- a/src/hapi/runs.py +++ b/src/hapi/runs.py @@ -79,6 +79,12 @@ class DistributedRun: `river_geometry` to identify them, checked here rather than in the routing loop. flow_path_length: Flow-path length raster, read only by :meth:`~hapi.rrm.distrrm.DistributedRRM.route_maxbas_by_path_length`. + keep_state_variables: Whether to allocate the per-cell state array. It is + `(rows, cols, time, 5)` -- as much memory as every other result field combined -- + and nothing but `save_results` and `plot_distributed_results` reads it, so a run + that will not look at it can halve its peak allocation. Defaults to True, which is + what every existing caller got; `Calibration` turns it off, because it runs the + model once per trial vector and never reads the states. """ period: SimulationPeriod @@ -89,6 +95,7 @@ class DistributedRun: river_geometry: RiverGeometry | None = None skip_hydraulic_cells: bool = False flow_path_length: np.ndarray | None = None + keep_state_variables: bool = True def __post_init__(self): """Check the inputs agree with each other and with the grid. @@ -160,6 +167,7 @@ def from_model( needs_flow_direction: bool = True, with_river_geometry: bool = False, skip_hydraulic_cells: bool = False, + keep_state_variables: bool = True, ) -> DistributedRun: """Narrow a built catchment into a validated distributed run. @@ -174,6 +182,9 @@ def from_model( reads it. with_river_geometry: Carry the river geometry through, for the flood path. skip_hydraulic_cells: Leave the river cells to a hydraulic model. + keep_state_variables: Allocate the per-cell state array. False halves the run's + peak memory at the cost of `save_results` / `plot_distributed_results` + options 4 to 8. Returns: DistributedRun: The validated inputs. @@ -218,6 +229,7 @@ def from_model( river_geometry=geometry, skip_hydraulic_cells=skip_hydraulic_cells, flow_path_length=getattr(model, "flow_path_length_arr", None), + keep_state_variables=keep_state_variables, ) diff --git a/tests/rrm/catchment/test_run_narrowing.py b/tests/rrm/catchment/test_run_narrowing.py index eecda2e5..3db2cc1c 100644 --- a/tests/rrm/catchment/test_run_narrowing.py +++ b/tests/rrm/catchment/test_run_narrowing.py @@ -251,3 +251,94 @@ def test_a_record_that_does_not_span_the_period_is_refused(self, built: Catchmen with pytest.raises(ValueError, match="the run is positional"): LumpedRun.from_model(built) + + +class TestStateVariablesAreOptional: + """The per-cell state array is diagnostic, and half the run's memory.""" + + def test_dropping_them_halves_the_result_arrays(self, built: Catchment): + """Test that opting out actually removes the allocation. + + Test scenario: + `state_variables` is `(rows, cols, time, 5)` -- as much memory as every other + result field combined -- and only `save_results` and `plot_distributed_results` + read it. A run that will not look at it should not pay for it. + """ + kept = Wrapper.run_muskingum(DistributedRun.from_model(built)) + dropped = Wrapper.run_muskingum( + DistributedRun.from_model(built, keep_state_variables=False) + ) + + assert kept.state_variables is not None, "kept by default" + assert dropped.state_variables is None, "dropped when asked" + + def size(results): + fields = ( + results.quz, + results.qlz, + results.quz_routed, + results.qlz_translated, + results.q_total, + ) + total = sum(a.nbytes for a in fields) + if results.state_variables is not None: + total += results.state_variables.nbytes + return total + + assert size(dropped) * 2 == size(kept), ( + f"the states are half the footprint; {size(dropped)} against {size(kept)}" + ) + + @pytest.mark.parametrize( + "field", ["quz", "qlz", "quz_routed", "qlz_translated", "q_total"] + ) + def test_the_discharge_is_unchanged_either_way(self, built: Catchment, field: str): + """Test that dropping the states changes nothing about the discharge. + + Test scenario: + The states are written but never read by the routing, so removing the allocation + must be invisible in the results. If it is not, something was reading them. + + Args: + field: The result field being compared. + """ + kept = Wrapper.run_muskingum(DistributedRun.from_model(built)) + dropped = Wrapper.run_muskingum( + DistributedRun.from_model(built, keep_state_variables=False) + ) + + np.testing.assert_array_equal( + getattr(kept, field), + getattr(dropped, field), + err_msg=f"{field} must not depend on whether the states were kept", + ) + + def test_a_state_option_says_why_it_cannot_be_saved(self, built: Catchment): + """Test that asking for a state option on such a run names the switch. + + Test scenario: + Without the guard this failed on `None` inside a slice, several frames from the + option that asked for it and naming nothing the caller controls. + """ + built.results = Wrapper.run_muskingum( + DistributedRun.from_model(built, keep_state_variables=False) + ) + + with pytest.raises(ValueError, match="keep_state_variables"): + built.plot_distributed_results("2009-01-01", "2009-01-05", option=4) + + def test_a_discharge_option_still_works_without_them(self, built: Catchment): + """Test that the guard only fires for the options that need the states. + + Test scenario: + Binding the states once at the top of the method would have made a + discharge-only plot raise on a run that never needed them. + """ + built.results = Wrapper.run_muskingum( + DistributedRun.from_model(built, keep_state_variables=False) + ) + + assert built.results.q_total is not None, "the discharge options read this" + # option 1 is total discharge; it must not consult the states at all + arr = built.results.q_total[:, :, 0:2] + assert arr.shape[2] == 2, "a discharge slice works with no states allocated" From 5d9fb6b760b3e134ce4754d9f0fab17f9fe1a810 Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Thu, 10 Sep 2026 21:10:17 +0200 Subject: [PATCH 17/54] refactor(results)!: move the plotting and saving off Catchment `Catchment` is a builder: it assembles inputs and hands them to the run layer. It also carried four methods that do the opposite -- take finished results and turn them into figures and files. Three of them, 332 lines, were the only reason the builder imported a plotting stack at all, and the only reason it held `anim` and `_animation_glyph`, two of its nineteen attributes. They read result arrays. They now live on the object that holds those arrays: * `plot_distributed_results` -> `SimulationResults.animate` * `save_animation` -> `SimulationResults.save_animation` * `save_results` -> `SimulationResults.save` `catchment.py` drops from 1,760 to 1,428 lines and no longer imports cleopatra, `DatasetCollection` or `matplotlib.animation`; `Catchment.__init__` goes from 19 attributes to 17. The methods need the calendar to index the arrays by, the grid to mask them with and the drivers to animate beside them, so `SimulationResults` now carries the `DistributedRun` or `LumpedRun` that produced it. That is provenance the object wanted anyway -- the arrays are not interpretable without it -- and it is the one part of this that is a design decision rather than a move. Results built by hand rather than by a run say what is missing instead of failing on `None`. cleopatra is imported *inside* `animate`, not at module scope. `hapi.results` is what every engine imports, so a top-level import would have put matplotlib in the path of every model run -- worse than the arrangement it replaced. `test_running_a_model_does_not_import_a_plotting_stack` holds it there. Three things fell out of the move rather than being aimed at: 1. `save` reads rasters-or-CSV off `routing` instead of the caller's `spatial_resolution`, so the choice is a property of the results rather than something restated at the call site. 2. The CSV branch built its index with a hard-coded `freq="D"`, so an hourly lumped run wrote a daily index against hourly values. It uses the run's own calendar now. 3. `gauges` was a `bool` that reached back onto the catchment for `GaugesTable`. It takes the table itself, which is what let the method stop knowing about catchments at all. `plot_hydrograph` deliberately stayed on `Catchment`. It reads no result array -- it plots `Qsim` against the observed gauge record, and `Qsim`, `metrics` and `QGauges` are analysis products, not run output. `SimulationResults` is now documented in the API reference, which it was not before. 602 tests pass plus 17 plot tests and 36 doctests; mypy clean. BREAKING CHANGE: `Catchment.plot_distributed_results`, `.save_animation` and `.save_results` are gone. Call them on the results a run returns: `model.results.animate(start, end, option=1)`, `model.results.save_animation(path, fps=2)` and `model.results.save(path, result=1, flow_acc_path=...)`. `save`'s first positional argument is now `path`, not `flow_acc_path`, and its date arguments are `start` and `end`. `animate`'s `gauges` takes the gauge table (`model.GaugesTable`) rather than a `bool`. `Catchment.anim` and `Catchment.STATE_VARIABLES` moved to `hapi.results`. --- docs/api/results.md | 50 ++ docs/examples/distributed-model-calib.md | 17 +- docs/examples/lumped-model-run.md | 2 +- docs/examples/run-configuration.md | 2 +- ...libration-deap-multiobjective-NSE-NSEHF.py | 2 +- ...alibration-deap-multiobjective-NSE-RMSE.py | 2 +- .../coello-lumped-model-calibration-deap.py | 2 +- .../coello-lumped-model-calibration.py | 2 +- .../coello-distributed-model-run-netcdf.py | 4 +- .../run/coello-lumped-model-run-maxbas.py | 2 +- .../coello/run/coello-lumped-model-run.py | 2 +- .../Jiboa-distributed-model-muskingum-lake.py | 10 +- mkdocs.yml | 1 + src/hapi/catchment.py | 342 +----------- src/hapi/config.py | 2 +- src/hapi/results.py | 518 +++++++++++++++++- src/hapi/rrm/distrrm.py | 4 + src/hapi/run.py | 8 +- src/hapi/runs.py | 4 +- src/hapi/wrapper.py | 5 +- tests/calibration/lumped_calibration.py | 2 +- tests/rrm/catchment/test_config.py | 2 +- .../catchment/test_e2e_coello_from_netcdf.py | 8 +- tests/rrm/catchment/test_fw1_output_fields.py | 14 +- tests/rrm/catchment/test_plot_animation.py | 89 ++- tests/rrm/catchment/test_rrm_catchment.py | 4 +- tests/rrm/catchment/test_run_narrowing.py | 4 +- .../catchment/test_run_results_coupling.py | 29 + .../test_save_results_distributed.py | 24 +- tests/rrm/catchment/test_wrapper_with_lake.py | 2 +- tests/run/distributed_mode_run.py | 18 +- tests/run/lumped_run.py | 2 +- 32 files changed, 735 insertions(+), 444 deletions(-) create mode 100644 docs/api/results.md diff --git a/docs/api/results.md b/docs/api/results.md new file mode 100644 index 00000000..edc986c0 --- /dev/null +++ b/docs/api/results.md @@ -0,0 +1,50 @@ +# Results + +Every `Run.*` entry point returns a `SimulationResults` and assigns it to `Catchment.results`. That +object is the only home for the arrays a run produced — the catchment carries no result attributes +of its own — and it is also what renders and writes them. + +## Reading a run + +```python +results = Run.run_distributed(model) # also assigned to model.results + +results.q_total # (rows, cols, time) total discharge +results.routing # RoutingKind.MUSKINGUM +results.run.period # the calendar the arrays are indexed by +``` + +`routing` is not decoration: it decides how one cell of `q_total` may be read. Under Muskingum the +discharge accumulates downstream, so a cell *is* the discharge at that cell. Under MAXBAS every cell +is routed straight to the outlet, so a cell is only that cell's contribution and the hydrograph is +the sum over the domain. Ask `results.outlet_shortcut_valid` rather than assuming. + +## Viewing and saving + +| Call | Does | +|---|---| +| `results.animate(start, end, option=1)` | Animates a result array or a driver over the grid. | +| `results.save_animation(path, fps=2)` | Writes the animation `animate` built. | +| `results.save(path, result=1, flow_acc_path=...)` | One GeoTIFF per step, or a CSV for a lumped run. | + +`animate` and `save` need the run behind the arrays — the calendar to index them by and the grid to +mask them with — which is why `SimulationResults` carries the `DistributedRun` or `LumpedRun` that +produced it. A results object built by hand rather than by a run says so instead of failing on +`None`. + +`save` chooses rasters or CSV from `routing`: a lumped run has no grid to write rasters on, and that +is a property of the results rather than something the caller restates. The raster branch needs +`flow_acc_path` because `FlowNetwork` keeps the accumulation *array* but not its projection, so the +georeferencing has to be read back from the file. + +Importing the run layer does not import matplotlib or cleopatra: `animate` imports them itself, so a +model run never pays for a plotting stack it does not use. + +`Catchment.plot_hydrograph` stayed on the catchment. It reads no result array — it compares `Qsim` +against the observed gauge record, which is an analysis input, not something a run produced. + +## SimulationResults +::: hapi.results.SimulationResults + +## RoutingKind +::: hapi.results.RoutingKind diff --git a/docs/examples/distributed-model-calib.md b/docs/examples/distributed-model-calib.md index f2bdddf5..22592b27 100644 --- a/docs/examples/distributed-model-calib.md +++ b/docs/examples/distributed-model-calib.md @@ -168,7 +168,8 @@ Coello.plot_hydrograph(plotstart, plotend, gaugei) ## 6-Animation - The best way to visualize a time series of distributed data is an animation. The `Catchment` object - has a `plot_distributed_results` method which animates any of the model results. + carries a `SimulationResults` object on `model.results`, whose `animate` method animates + any of them. The keyword arguments are forwarded to `cleopatra.glyphs.gridded.array_glyph.ArrayGlyph.animate`; see its documentation for the full list. @@ -178,7 +179,7 @@ cleopatra 0.30 moved the styling keywords onto typed group objects, so the colou `.boundary(bounds=...)`), the cell-value labels are `cells=CellValues(show=True, size=..., background_threshold=...)`, and the frame time-stamp is `frame_label=FrameLabel(location=[...], color=...)`. The gauge markers are built by Hapi itself -when `gauges=True`. +from the gauge table you pass as `gauges=`. `option` selects the variable to animate: @@ -199,11 +200,11 @@ from cleopatra.styling.scaling import ColorScaling plotstart = "2009-01-01" plotend = "2009-04-20" -anim = Coello.plot_distributed_results( +anim = Coello.results.animate( plotstart, plotend, option=1, - gauges=True, + gauges=Coello.GaugesTable, figsize=(9, 9), ticks_spacing=5, interval=200, @@ -221,7 +222,7 @@ anim = Coello.plot_distributed_results( system. ```python -Coello.save_animation("results/anim.gif", fps=2) +Coello.results.save_animation("results/anim.gif", fps=2) ``` ## 7-Save the result into rasters @@ -232,12 +233,12 @@ start = "2009-01-01" end = "2010-04-20" prefix = "Qtot_" -Coello.save_results( - FlowAccPath, +Coello.results.save( + path="results/", + flow_acc_path=FlowAccPath, result=1, start=start, end=end, - path="results/", prefix=prefix, ) ``` diff --git a/docs/examples/lumped-model-run.md b/docs/examples/lumped-model-run.md index b5c9442f..40327329 100644 --- a/docs/examples/lumped-model-run.md +++ b/docs/examples/lumped-model-run.md @@ -110,5 +110,5 @@ start = "2009-01-01" end = "2010-04-20" Path = SaveTo + "Results-Lumped-Model" + str(dt.datetime.now())[0:10] + ".txt" -Coello.save_results(result=5, start=start, end=end, path=Path) +Coello.results.save(result=5, start=start, end=end, path=Path) ``` diff --git a/docs/examples/run-configuration.md b/docs/examples/run-configuration.md index 015ea830..a4de0628 100644 --- a/docs/examples/run-configuration.md +++ b/docs/examples/run-configuration.md @@ -147,7 +147,7 @@ itself consume remain reachable — `outputs` above all: ```python outputs = Coello.config.outputs save_to = (outputs.results_dir if outputs is not None else None) or "" -Coello.save_results( +Coello.results.save( flow_acc_path=Coello.config.flow_network.flow_accumulation, result=1, path=save_to, diff --git a/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-NSEHF.py b/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-NSEHF.py index 7248b8c7..b7465624 100644 --- a/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-NSEHF.py +++ b/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-NSEHF.py @@ -201,4 +201,4 @@ def distance(individual): + str(dt.datetime.now())[0:10] + ".txt" ) -Coello.model.save_results(result=5, start=StartDate, end=EndDate, path=Path) +Coello.model.results.save(result=5, start=StartDate, end=EndDate, path=Path) diff --git a/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-RMSE.py b/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-RMSE.py index 03d18d2e..c33dc6ba 100644 --- a/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-RMSE.py +++ b/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-RMSE.py @@ -204,4 +204,4 @@ def distance(individual): + str(dt.datetime.now())[0:10] + ".txt" ) -Coello.model.save_results(result=5, start=StartDate, end=EndDate, path=Path) +Coello.model.results.save(result=5, start=StartDate, end=EndDate, path=Path) diff --git a/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap.py b/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap.py index 4ba3560f..47263942 100644 --- a/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap.py +++ b/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap.py @@ -193,4 +193,4 @@ def distance(individual): Path = ( Path + f"{Coello.name}-results-lumped-model" + str(dt.datetime.now())[0:10] + ".txt" ) -Coello.model.save_results(result=5, start=StartDate, end=EndDate, path=Path) +Coello.model.results.save(result=5, start=StartDate, end=EndDate, path=Path) diff --git a/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration.py b/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration.py index 5501861c..8ca38b41 100644 --- a/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration.py +++ b/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration.py @@ -148,4 +148,4 @@ EndDate = "2010-04-20" Path = Path + "Results-Lumped-Model" + str(dt.datetime.now())[0:10] + ".txt" -Coello.model.save_results(result=5, start=StartDate, end=EndDate, path=Path) +Coello.model.results.save(result=5, start=StartDate, end=EndDate, path=Path) diff --git a/examples/hydrological-model/coello/run/coello-distributed-model-run-netcdf.py b/examples/hydrological-model/coello/run/coello-distributed-model-run-netcdf.py index 6156af79..70940945 100644 --- a/examples/hydrological-model/coello/run/coello-distributed-model-run-netcdf.py +++ b/examples/hydrological-model/coello/run/coello-distributed-model-run-netcdf.py @@ -84,14 +84,14 @@ print(f"WB= {Coello.metrics.loc['WB', gauge_id]:.2f}") # %% Save the routed discharge to rasters, one per time step -# Both paths come from the configuration rather than being restated here: `save_results` +# Both paths come from the configuration rather than being restated here: `results.save` # re-reads the flow-accumulation raster for georeferencing (FlowNetwork keeps only the arrays, # not the source path), and `outputs.results_dir` says where the rasters go. The block is # optional, so a configuration without one writes beside the script rather than failing on a # missing attribute. outputs = Coello.config.outputs save_to = (outputs.results_dir if outputs is not None else None) or "" -Coello.save_results( +Coello.results.save( flow_acc_path=Coello.config.flow_network.flow_accumulation, result=1, path=save_to, diff --git a/examples/hydrological-model/coello/run/coello-lumped-model-run-maxbas.py b/examples/hydrological-model/coello/run/coello-lumped-model-run-maxbas.py index 5f959441..784768d7 100644 --- a/examples/hydrological-model/coello/run/coello-lumped-model-run-maxbas.py +++ b/examples/hydrological-model/coello/run/coello-lumped-model-run-maxbas.py @@ -64,5 +64,5 @@ EndDate = "2010-04-20" path = f"{SaveTo}/{Coello.name}Results-Lumped-Model_{str(dt.datetime.now())[0:10]}.txt" -Coello.save_results(result=5, start=StartDate, end=EndDate, path=path) +Coello.results.save(result=5, start=StartDate, end=EndDate, path=path) print(f"results written to : {path}") diff --git a/examples/hydrological-model/coello/run/coello-lumped-model-run.py b/examples/hydrological-model/coello/run/coello-lumped-model-run.py index ea822730..7dff278e 100644 --- a/examples/hydrological-model/coello/run/coello-lumped-model-run.py +++ b/examples/hydrological-model/coello/run/coello-lumped-model-run.py @@ -64,5 +64,5 @@ EndDate = "2010-04-20" path = f"{SaveTo}/Results-Lumped-Model_{str(dt.datetime.now())[0:10]}.txt" -Coello.save_results(result=5, start=StartDate, end=EndDate, path=path) +Coello.results.save(result=5, start=StartDate, end=EndDate, path=path) print(f"results written to : {path}") diff --git a/examples/hydrological-model/jiboa/Jiboa-distributed-model-muskingum-lake.py b/examples/hydrological-model/jiboa/Jiboa-distributed-model-muskingum-lake.py index a5b7603d..2fa2aa04 100644 --- a/examples/hydrological-model/jiboa/Jiboa-distributed-model-muskingum-lake.py +++ b/examples/hydrological-model/jiboa/Jiboa-distributed-model-muskingum-lake.py @@ -177,7 +177,7 @@ """ Animate the distributed results. -plot_distributed_results animates the time series of the meteorological +SimulationResults.animate animates the time series of the meteorological inputs and the results calculated by the model, like the total discharge, upper zone and lower zone discharge, and the state variables. The keyword arguments are forwarded to @@ -189,7 +189,7 @@ plotstart = "2012-07-20" plotend = "2012-08-20" -Anim = Jiboa.plot_distributed_results( +Anim = Jiboa.results.animate( plotstart, plotend, figsize=(8, 8), @@ -197,19 +197,19 @@ cells=CellValues(show=False, background_threshold=160), ticks_spacing=10, interval=10, - gauges=False, + gauges=None, cmap="inferno", frame_label=FrameLabel(location=[0.6, 0.8]), color=ColorScaling.power(gamma=0.08), ) # %% Path = save_to + "anim.mov" -Jiboa.save_animation(Path, fps=2) +Jiboa.results.save_animation(Path, fps=2) # %% Save Results start_date = "2012-07-20" end_date = "2012-08-20" Path = save_to + "Lumped_Parameters_" + str(dt.datetime.now())[0:10] + "_" -Jiboa.save_results( +Jiboa.results.save( result=1, start=start_date, end=end_date, path=Path, flow_acc_path=flow_acc_path ) diff --git a/mkdocs.yml b/mkdocs.yml index 4dd268d5..af93f7eb 100644 --- a/mkdocs.yml +++ b/mkdocs.yml @@ -110,6 +110,7 @@ nav: - api/inputs.md - api/routing.md - api/run.md + - api/results.md - api/wrapper.md - api/distrrm.md - api/hbv.md diff --git a/src/hapi/catchment.py b/src/hapi/catchment.py index f75c71b2..47692165 100644 --- a/src/hapi/catchment.py +++ b/src/hapi/catchment.py @@ -21,7 +21,7 @@ from collections.abc import Iterator from contextlib import contextmanager from pathlib import Path -from typing import TYPE_CHECKING, Any, Self +from typing import TYPE_CHECKING, Self import matplotlib.dates as dates import matplotlib.pyplot as plt @@ -29,10 +29,8 @@ import pandas as pd import statista.descriptors as metrics import yaml -from cleopatra.glyphs.gridded.array_glyph import ArrayGlyph, PointOverlay from loguru import logger from pyramids.dataset import Dataset -from pyramids.dataset import DatasetCollection as Datacube from pyramids.feature import FeatureCollection from hapi.conceptual import ConceptualModelSetup, ParameterSet @@ -51,11 +49,8 @@ from hapi.rrm.hbv_bergestrom92 import HBVBergestrom92 if TYPE_CHECKING: - import matplotlib.animation - from hapi.rrm.base_model import BaseConceptualModel -STATE_VARIABLES = ["SP", "SM", "UZ", "LZ", "WC"] CONVERSION_FACTOR = (1000 * 24 * 60 * 60) / (1000**2) #: (snow, maxbas) -> how many parameters the conceptual model reads in that configuration. PARAMETER_COUNTS = { @@ -223,30 +218,6 @@ def _name_the_path(path) -> Iterator[None]: raise FileNotFoundError(f"{exc} (path: {path})") from exc -def _require_state_variables(results: SimulationResults) -> np.ndarray: - """Return the per-cell state array, or say why it is absent. - - It is `(rows, cols, time, 5)` -- as much memory as every other result field combined -- so a - run can be asked not to keep it. Only these plotting and saving options read it, so the - error belongs here, naming the switch rather than failing on `None` inside a slice. - - Args: - results: The run's results. - - Returns: - np.ndarray: The state array. - - Raises: - ValueError: The run was asked not to keep the states. - """ - if results.state_variables is None: - raise ValueError( - "this run did not keep the state variables, so no state option can be plotted or " - "saved; run it with keep_state_variables=True (the default) if you need them" - ) - return results.state_variables - - class Catchment: """Catchment for reading meteorological/spatial inputs and running the model. @@ -354,13 +325,11 @@ def __init__( #: The five hydraulic rasters the flood model reads, once `read_river_geometry` has #: run. Absent-or-complete: they are checked against each other as they are read. self.river_geometry: RiverGeometry | None = None - #: Everything one run produced, replaced wholesale by the next run. The seven - #: result arrays below are read-only properties forwarding to it, so `model.results.q_total` - #: still reads as it always did while the run layer owns the arrays. `None` until - #: a `Run.*` entry point has been called. + #: Everything one run produced, replaced wholesale by the next run -- the arrays, + #: the routing that made them, and the methods that render and write them + #: (`model.results.animate(...)`, `model.results.save(...)`). `None` until a `Run.*` + #: entry point has been called. self.results: SimulationResults | None = None - self.anim: matplotlib.animation.FuncAnimation | None = None - self._animation_glyph: ArrayGlyph | None = None self.Qsim: np.ndarray | None = None self.metrics: pd.DataFrame | None = None #: The configuration this model was built from, when it came from @@ -1314,307 +1283,6 @@ def plot_hydrograph( return fig, ax - def plot_distributed_results( - self, - start: str | dt.datetime, - end: str | dt.datetime, - fmt: str = "%Y-%m-%d", - option: int = 1, - gauges: bool = False, - **kwargs: Any, - ): - """Animate distributed model results or meteorological inputs. - - Creates an animation of the time series of meteorological inputs - or model results (discharge, state variables) over the spatial - domain. Cells outside the catchment domain are masked on a copy of - the data, so the model arrays stored on the instance are never - modified. The animation title defaults to the selected variable's - name; an explicit `title=` keyword argument overrides it. - - Args: - start (str): Starting date for the animation. - end (str): End date for the animation. - fmt (str, optional): Format of the given date. Default - is "%Y-%m-%d". - option (int, optional): Variable to animate. Options are: - 1 - Total discharge, 2 - Upper zone discharge, - 3 - Ground water, 4 - Snow pack, 5 - Soil moisture, - 6 - Upper zone, 7 - Lower zone, 8 - Water content, - 9 - Precipitation, 10 - ET, 11 - Temperature. - Default is 1. - gauges (bool, optional): Whether to plot gauge locations - on the animation. Default is False. - **kwargs: Additional keyword arguments passed to - `ArrayGlyph.animate`. Loose styling keywords still - accepted: title (str), title_size (int), cmap (str), - vmin (float), vmax (float), interval (int), - figsize (tuple), cell_value_text_colors (tuple), - ticks_spacing (int), cbar_label (str), - cbar_label_size (int), cbar_length (float), - cbar_orientation (str). - Styling that cleopatra 0.30 moved onto typed group - objects is passed as those objects instead: - color=`ColorScaling` (was color_scale / gamma / - bounds / midpoint), cells=`CellValues` (was - display_cell_value / num_size / - background_color_threshold), - contour=`Contour` (was levels), - data_style=`DataStyle` (was style / hillshade), - frame_label=`FrameLabel` (was label_location / - label_color / text_loc). See - `cleopatra.glyphs.gridded.array_glyph.ArrayGlyph.animate` - for the full list. - - Returns: - matplotlib.animation.FuncAnimation: The animation object. - - Raises: - ValueError: If `option` is not between 1 and 11. - """ - start = dt.datetime.strptime(start, fmt) - end = dt.datetime.strptime(end, fmt) - - start_i = np.nonzero(self.period.date_index == start)[0][0] - end_i = np.nonzero(self.period.date_index == end)[0][0] - - if option == 1: - arr = self.results.q_total[:, :, start_i:end_i] - title = "Total Discharge" - elif option == 2: - arr = self.results.quz_routed[:, :, start_i:end_i] - title = "Surface Flow" - elif option == 3: - arr = self.results.qlz_translated[:, :, start_i:end_i] - title = "Ground Water Flow" - elif option == 4: - arr = _require_state_variables(self.results)[:, :, start_i:end_i, 0] - title = "Snow Pack" - elif option == 5: - arr = _require_state_variables(self.results)[:, :, start_i:end_i, 1] - title = "Soil Moisture" - elif option == 6: - arr = _require_state_variables(self.results)[:, :, start_i:end_i, 2] - title = "Upper Zone" - elif option == 7: - arr = _require_state_variables(self.results)[:, :, start_i:end_i, 3] - title = "Lower Zone" - elif option == 8: - arr = _require_state_variables(self.results)[:, :, start_i:end_i, 4] - title = "Water Content" - elif option == 9: - arr = self.meteo.precipitation[:, :, start_i:end_i] - title = "Precipitation" - elif option == 10: - arr = self.meteo.evapotranspiration[:, :, start_i:end_i] - title = "ET" - elif option == 11: - arr = self.meteo.temperature[:, :, start_i:end_i] - title = "Temperature" - else: - raise ValueError("Plotting options are from 1 to 11") - - # mask the no-data cells on a copy so plotting never mutates the model - # result arrays stored on the instance - arr = arr.copy() - arr[np.isnan(self.flow_network.flow_acc_arr), :] = np.nan - - time = self.period.date_index[start_i:end_i] - - if gauges: - # animate expects a 3-column array: [value to display, cell row, cell column]. - # cleopatra 0.30 stopped accepting a bare array; it must be wrapped in a - # PointOverlay, which also carries the marker/label styling. - kwargs["points"] = PointOverlay( - self.GaugesTable[["id", "cell_row", "cell_col"]].to_numpy() - ) - - # animate iterates over the first dimension, so move the time axis to the front - array = ArrayGlyph(np.moveaxis(arr, -1, 0)) - # the option title is a default; an explicit title= kwarg wins - kwargs.setdefault("title", title) - anim = array.animate(time, **kwargs) - - self._animation_glyph = array - self.anim = anim - - return anim - - def save_animation(self, path: str, fps: int = 2): - """Save the animation created by `plot_distributed_results`. - - The output format is determined by the file extension. GIF uses - PillowWriter; mov/avi/mp4 require FFmpeg to be installed. - - Args: - path (str): Output file path. The extension determines the - format (gif, mov, avi, or mp4). - fps (int, optional): Frames per second. Default is 2. - - Raises: - ValueError: If `plot_distributed_results` has not been called - yet, or if the file format is not supported. - FileNotFoundError: If a video format is requested but FFmpeg - is not installed. - """ - if self._animation_glyph is None: - raise ValueError( - "There is no animation to save, call `plot_distributed_results` first" - ) - self._animation_glyph.save_animation(path, fps=fps) - - def save_results( - self, - flow_acc_path: str = "", - result: int = 1, - start: str | dt.datetime = "", - end: str | dt.datetime = "", - path: str = "", - prefix: str = "", - fmt: str = "%Y-%m-%d", - ): - """Save model results to raster files or CSV. - - For distributed mode, saves results as GeoTIFF rasters. For - lumped mode, saves results as a CSV file. - - Args: - flow_acc_path (str, optional): Path to the flow - accumulation raster (required for distributed mode). - Default is "". - result (int, optional): Type of result to save: - 1 - Total discharge, 2 - Upper zone discharge, - 3 - Lower zone discharge, 4 - Snow pack, - 5 - Soil moisture, 6 - Upper zone, 7 - Lower zone, - 8 - Water content. For lumped mode, 5 saves all - variables. Default is 1. - start (str | dt.datetime, optional): Start date for the - output period. A string is parsed with `fmt`; a datetime - is used as it is. - If empty, uses the first index. Default is "". - end (str | dt.datetime, optional): End date for the output - period. See `start`. If - empty, uses the last index. Default is "". - path (str, optional): Output directory (distributed, created - if it does not exist) or the CSV file itself (lumped). - Default is "", the working directory. - prefix (str, optional): Prefix for the output file - names. Default is "". - fmt (str, optional): Date format for parsing `start` and - `end`. Default is "%Y-%m-%d". - - Raises: - Exception: If `flow_acc_path` is not provided in - distributed mode. - TypeError: If `path` is not a string. `outputs.results_dir` - is optional in a run configuration, so a caller - forwarding it can hold None. - ValueError: If `result` is not a valid option. - """ - if not isinstance(path, str): - raise TypeError( - f"path must be a string naming a directory (distributed) or a file " - f"(lumped), got {type(path).__name__}" - ) - - if start == "": - start = self.period.date_index[0] - elif isinstance(start, str): - start = dt.datetime.strptime(start, fmt) - - if end == "": - end = self.period.date_index[-1] - elif isinstance(end, str): - end = dt.datetime.strptime(end, fmt) - - start_i = np.nonzero(self.period.date_index == start)[0][0] - end_i = np.nonzero(self.period.date_index == end)[0][0] + 1 - - if self.spatial_resolution == "distributed": - if flow_acc_path == "": - raise Exception( - "Please enter the FlowAccPath parameter to the saveResults method" - ) - - src = Dataset.read_file(flow_acc_path) - - if prefix == "": - prefix = "Result_" - - # `path` names a directory here, unlike the lumped branch below where it is the - # CSV itself. Joined rather than concatenated: the old `path + prefix` wrote - # `some/dirResult_2009-01-01.tif` for any directory given without a trailing - # separator, which is how a directory is normally written. - if path and not os.path.isdir(path): - os.makedirs(path, exist_ok=True) - names = [ - os.path.join(path, f"{prefix}{str(i)[:10]}.tif") - for i in self.period.date_index[start_i:end_i] - ] - if result == 1: - arr = self.results.q_total[:, :, start_i:end_i] - elif result == 2: - arr = self.results.quz_routed[:, :, start_i:end_i] - elif result == 3: - arr = self.results.qlz_translated[:, :, start_i:end_i] - elif result == 4: - arr = _require_state_variables(self.results)[:, :, start_i:end_i, 0] - elif result == 5: - arr = _require_state_variables(self.results)[:, :, start_i:end_i, 1] - elif result == 6: - arr = _require_state_variables(self.results)[:, :, start_i:end_i, 2] - elif result == 7: - arr = _require_state_variables(self.results)[:, :, start_i:end_i, 3] - elif result == 8: - arr = _require_state_variables(self.results)[:, :, start_i:end_i, 4] - else: - raise ValueError( - f" The result parameter takes a value between 1 and 8, given: {result}" - ) - - # from_dataset is pyramids' named constructor for an in-memory - # scaffold off a template raster; the bare Datacube(src, time_length=) - # form it replaced is kept only as a legacy fallback upstream. - cube = Datacube.from_dataset(src, arr.shape[2]) - arr = np.moveaxis(arr, -1, 0) - cube.values = arr - cube.to_file(names) - else: - ind = pd.date_range(start, end, freq="D") - data = pd.DataFrame(index=ind) - - data["date"] = ["'" + str(i)[:10] + "'" for i in data.index] - - if result == 1: - data["Qsim"] = self.Qsim[start_i:end_i] - data.to_csv(path, index=False, float_format="%.3f") - elif result == 2: - data["Quz"] = self.results.quz[start_i:end_i] - data.to_csv(path, index=False, float_format="%.3f") - elif result == 3: - data["Qlz"] = self.results.qlz[start_i:end_i] - data.to_csv(path, index=False, float_format="%.3f") - elif result == 4: - data[STATE_VARIABLES] = _require_state_variables(self.results)[ - start_i:end_i, : - ] - data.to_csv(path, index=False, float_format="%.3f") - elif result == 5: - data["Qsim"] = self.Qsim[start_i:end_i] - data["Quz"] = self.results.quz[start_i:end_i] - data["Qlz"] = self.results.qlz[start_i:end_i] - data[STATE_VARIABLES] = _require_state_variables(self.results)[ - start_i:end_i, : - ] - data.to_csv(path, index=False, float_format="%.3f") - else: - raise ValueError( - f"in lumped mode the result parameter takes a value between 1 and 5, " - f"given: {result}" - ) - - logger.debug("Data is saved successfully") - class Lake: """Lake simulation using a lumped model with a rating curve. diff --git a/src/hapi/config.py b/src/hapi/config.py index e5e5e07e..28a9184d 100644 --- a/src/hapi/config.py +++ b/src/hapi/config.py @@ -460,7 +460,7 @@ class OutputsConfig(BaseModel): """Where to write results after the run. Attributes: - results_dir: Folder `save_results` writes into. + results_dir: Folder `SimulationResults.save` writes into. """ model_config = _STRICT diff --git a/src/hapi/results.py b/src/hapi/results.py index 47b5c799..6cbcaedb 100644 --- a/src/hapi/results.py +++ b/src/hapi/results.py @@ -1,4 +1,4 @@ -"""The arrays a model run produces, and the routing that produced them. +"""The arrays a model run produces, the routing that produced them, and how to view them. Running a model used to leave its output as nine separate attributes on the :class:`~hapi.catchment.Catchment` it was handed, with a private boolean recording which @@ -10,14 +10,73 @@ :class:`SimulationResults` holds them together instead, with the routing scheme as a field. A run assigns one to `Catchment.results`, and that is the only place the arrays live -- read them as `model.results.q_total`. The catchment carries no result attributes of its own. + +It also renders and writes them. :meth:`~SimulationResults.animate`, +:meth:`~SimulationResults.save_animation` and :meth:`~SimulationResults.save` used to sit on +`Catchment` -- around three hundred lines of matplotlib, cleopatra and pyramids on the object +whose job is to *assemble inputs*, and the only reason a builder imported a plotting stack at +all. They read result arrays and the run that produced them and touch nothing a catchment +alone holds, so they live here, beside the arrays they render. + +`Catchment.plot_hydrograph` deliberately stayed behind: it reads no result array at all. It +compares `Qsim` against the observed gauge record, which is an analysis input this object has +no claim to. """ from __future__ import annotations -from dataclasses import dataclass +import datetime as dt +import os +from dataclasses import dataclass, field from enum import Enum +from typing import TYPE_CHECKING, Any import numpy as np +import pandas as pd +from loguru import logger +from pyramids.dataset import Dataset +from pyramids.dataset import DatasetCollection as Datacube + +from hapi.runs import DistributedRun, LumpedRun + +if TYPE_CHECKING: + import matplotlib.animation + from cleopatra.glyphs.gridded.array_glyph import ArrayGlyph + + from hapi.period import SimulationPeriod + +#: The five per-cell states, in the order the last axis of `state_variables` carries them. +STATE_VARIABLES = ["SP", "SM", "UZ", "LZ", "WC"] + +#: `animate` option -> (the attribute it reads, the default title). Options 1-3 are result +#: arrays, 4-8 are slices of the state array, and 9-11 are the meteorological drivers the run +#: was given -- animated on the same grid, which is why they live on the same switch. +_ANIMATION_OPTIONS: dict[int, tuple[str, str]] = { + 1: ("q_total", "Total Discharge"), + 2: ("quz_routed", "Surface Flow"), + 3: ("qlz_translated", "Ground Water Flow"), + 4: ("state:0", "Snow Pack"), + 5: ("state:1", "Soil Moisture"), + 6: ("state:2", "Upper Zone"), + 7: ("state:3", "Lower Zone"), + 8: ("state:4", "Water Content"), + 9: ("meteo:precipitation", "Precipitation"), + 10: ("meteo:evapotranspiration", "ET"), + 11: ("meteo:temperature", "Temperature"), +} + +#: `save` option -> the result attribute it writes, for a distributed run. The state options +#: name the slice of `state_variables` rather than a field of their own. +_RASTER_OPTIONS: dict[int, str] = { + 1: "q_total", + 2: "quz_routed", + 3: "qlz_translated", + 4: "state:0", + 5: "state:1", + 6: "state:2", + 7: "state:3", + 8: "state:4", +} class RoutingKind(Enum): @@ -47,7 +106,7 @@ class RoutingKind(Enum): @dataclass class SimulationResults: - """The arrays one model run produced, and the routing that produced them. + """The arrays one model run produced, the routing that produced them, and their views. Built by the run layer and assigned to `Catchment.results`. Mutable, because the run fills it in stages: the per-cell model writes :attr:`quz`, :attr:`qlz` and @@ -61,8 +120,8 @@ class SimulationResults: state_variables: `(rows, cols, time, 5)` state array, the states being `[sp, sm, uz, lz, wc]`. For a lumped run, `(time, 5)`. `None` when a distributed run was asked not to keep them -- it is five times the size of every other field - put together and nothing but `save_results` and `plot_distributed_results` reads - it, so a run that will not look at it need not pay for it. See + put together and nothing but :meth:`save` and :meth:`animate` reads it, so a run + that will not look at it need not pay for it. See :attr:`~hapi.runs.DistributedRun.keep_state_variables`. quz_routed: Upper-zone discharge after routing. `None` until a routing step runs. qlz_translated: Lower-zone discharge after translation. `None` until then. @@ -72,6 +131,13 @@ class SimulationResults: domain and set it directly; the Muskingum paths leave it `None` for :meth:`~hapi.catchment.Catchment.extract_discharge` to read off the outlet cell, which needs the gauge table the engine does not have. + run: The validated inputs these arrays came from, carried as provenance. It is what + makes the arrays interpretable on their own: the calendar to index them by, the + grid to mask them with, and the drivers the animation options can show beside + them. `None` only for a results object built by hand rather than by a run, in + which case the presentation methods say so rather than failing on `None`. + anim: The animation :meth:`animate` last built, or `None`. Not a constructor + argument. Examples: - A freshly run, unrouted set knows it is not yet interpretable at the outlet: @@ -102,6 +168,18 @@ class SimulationResults: >>> muskingum.outlet_shortcut_valid, maxbas.outlet_shortcut_valid (True, False) + ``` + - Arrays with no run behind them say what is missing rather than failing on `None`: + ```python + >>> import numpy as np + >>> from hapi.results import RoutingKind, SimulationResults + >>> cube = np.zeros((2, 3, 4), dtype="float32") + >>> orphan = SimulationResults(RoutingKind.MUSKINGUM, cube, cube, None) + >>> orphan.save(path="out") + Traceback (most recent call last): + ... + ValueError: these results carry no run... + ``` """ @@ -113,6 +191,13 @@ class SimulationResults: qlz_translated: np.ndarray | None = None q_total: np.ndarray | None = None qout: np.ndarray | None = None + run: DistributedRun | LumpedRun | None = None + anim: matplotlib.animation.FuncAnimation | None = field( + default=None, init=False, repr=False + ) + # The glyph, not just the animation: cleopatra writes the file through the object that + # built the frames, so `save_animation` needs the glyph `animate` kept, not its return. + _animation_glyph: ArrayGlyph | None = field(default=None, init=False, repr=False) @property def outlet_shortcut_valid(self) -> bool: @@ -123,3 +208,426 @@ def outlet_shortcut_valid(self) -> bool: of a MAXBAS run under-reports the hydrograph, which is what this guards. """ return self.routing is not RoutingKind.MAXBAS + + # ------------------------------------------------------------------ # + # narrowing helpers + # ------------------------------------------------------------------ # + + def _require_run(self) -> DistributedRun | LumpedRun: + """Return the run behind these arrays, or say that there is none. + + Returns: + DistributedRun | LumpedRun: The run that produced these results. + + Raises: + ValueError: The results were built by hand rather than by a run. + """ + if self.run is None: + raise ValueError( + "these results carry no run, so there is no calendar to index them by and " + "no grid to write them on; they were built directly rather than by a " + "`Run.*` entry point" + ) + return self.run + + def _require_distributed_run(self) -> DistributedRun: + """Return the run as a distributed one, or say that it is not. + + Returns: + DistributedRun: The distributed run that produced these results. + + Raises: + ValueError: There is no run, or it was a lumped one, which has neither a grid + nor spatial drivers to render. + """ + run = self._require_run() + if not isinstance(run, DistributedRun): + raise ValueError( + "these results came from a lumped run, which has no grid to render or " + "write rasters from; use `save` to write them as a CSV instead" + ) + return run + + def _require_state_variables(self) -> np.ndarray: + """Return the per-cell state array, or say why it is absent. + + It is `(rows, cols, time, 5)` -- as much memory as every other result field combined + -- so a run can be asked not to keep it. Only these plotting and saving options read + it, so the error belongs here, naming the switch rather than failing on `None` inside + a slice. + + Returns: + np.ndarray: The state array. + + Raises: + ValueError: The run was asked not to keep the states. + """ + if self.state_variables is None: + raise ValueError( + "this run did not keep the state variables, so no state option can be " + "plotted or saved; run it with keep_state_variables=True (the default) if " + "you need them" + ) + return self.state_variables + + def _require_field(self, name: str) -> np.ndarray: + """Return a routed result field, or say which step has not run. + + Args: + name: The attribute to read. + + Returns: + np.ndarray: The field. + + Raises: + ValueError: The field is still `None` because no routing step has run. + """ + value: np.ndarray | None = getattr(self, name) + if value is None: + raise ValueError( + f"`{name}` is empty because no routing step has filled it; these results " + f"are {self.routing.value}" + ) + return value + + def _step_bounds( + self, + period: SimulationPeriod, + start: str | dt.datetime, + end: str | dt.datetime, + fmt: str, + inclusive: bool, + ) -> tuple[int, int]: + """Resolve two dates to positions in the run's calendar. + + Args: + period: The span the run covered. + start: First date, or `""` for the first step. + end: Last date, or `""` for the last step. + fmt: `strptime` format a string date is read with. + inclusive: Whether `end` itself is included in the range. + + Returns: + tuple[int, int]: The half-open `(start, end)` positions. + + Raises: + ValueError: A date is not a step of the run's calendar. + """ + index = period.date_index + if start == "": + start = index[0] + elif isinstance(start, str): + start = dt.datetime.strptime(start, fmt) + if end == "": + end = index[-1] + elif isinstance(end, str): + end = dt.datetime.strptime(end, fmt) + + for label, value in (("start", start), ("end", end)): + if not (index == value).any(): + raise ValueError( + f"{label} date {value} is not a step of this run, which covers " + f"{index[0]} to {index[-1]}" + ) + + start_i = int(np.nonzero(index == start)[0][0]) + end_i = int(np.nonzero(index == end)[0][0]) + (1 if inclusive else 0) + return start_i, end_i + + def _select(self, option: str, start_i: int, end_i: int) -> np.ndarray: + """Slice the array an option names out of the results or the run's drivers. + + Args: + option: Either an attribute name, `"state:"` for a slice of the state array, + or `"meteo:"` for one of the run's drivers. + start_i: First step. + end_i: One past the last step. + + Returns: + np.ndarray: The `(rows, cols, time)` slice. + """ + if option.startswith("state:"): + layer = int(option.split(":")[1]) + return self._require_state_variables()[:, :, start_i:end_i, layer] + if option.startswith("meteo:"): + run = self._require_distributed_run() + driver: np.ndarray = getattr(run.meteo, option.split(":")[1]) + return driver[:, :, start_i:end_i] + return self._require_field(option)[:, :, start_i:end_i] + + # ------------------------------------------------------------------ # + # presentation + # ------------------------------------------------------------------ # + + def animate( + self, + start: str | dt.datetime, + end: str | dt.datetime, + fmt: str = "%Y-%m-%d", + option: int = 1, + gauges: pd.DataFrame | None = None, + **kwargs: Any, + ) -> matplotlib.animation.FuncAnimation: + """Animate a result array or one of the run's drivers over the spatial domain. + + Cells outside the catchment domain are masked on a copy of the data, so the arrays + held here are never modified. The animation title defaults to the selected variable's + name; an explicit `title=` keyword argument overrides it. + + Args: + start: Starting date of the animation. + end: End date of the animation. + fmt: Format a string date is read with. Default is "%Y-%m-%d". + option: Variable to animate. 1 - Total discharge, 2 - Upper zone discharge, + 3 - Ground water, 4 - Snow pack, 5 - Soil moisture, 6 - Upper zone, + 7 - Lower zone, 8 - Water content, 9 - Precipitation, 10 - ET, + 11 - Temperature. Default is 1. + gauges: Gauge table to overlay, as `Catchment.GaugesTable`. It must carry `id`, + `cell_row` and `cell_col` columns. `None`, the default, draws no gauges. + This used to be a `bool` that reached back onto the catchment for the table; + the table is an analysis input, so it is passed in. + **kwargs: Additional keyword arguments passed to `ArrayGlyph.animate`. Loose + styling keywords still accepted: title (str), title_size (int), cmap (str), + vmin (float), vmax (float), interval (int), figsize (tuple), + cell_value_text_colors (tuple), ticks_spacing (int), cbar_label (str), + cbar_label_size (int), cbar_length (float), cbar_orientation (str). + Styling that cleopatra 0.30 moved onto typed group objects is passed as those + objects instead: color=`ColorScaling` (was color_scale / gamma / bounds / + midpoint), cells=`CellValues` (was display_cell_value / num_size / + background_color_threshold), contour=`Contour` (was levels), + data_style=`DataStyle` (was style / hillshade), frame_label=`FrameLabel` + (was label_location / label_color / text_loc). See + `cleopatra.glyphs.gridded.array_glyph.ArrayGlyph.animate` for the full list. + + Returns: + matplotlib.animation.FuncAnimation: The animation object, also kept on + :attr:`anim` so :meth:`save_animation` can write it. + + Raises: + ValueError: `option` is not between 1 and 11, the results carry no distributed + run, or a state option was asked for on a run that dropped the states. + """ + # cleopatra pulls in matplotlib, and this module is imported by the engines + # (`distrrm`, `wrapper`, `run`). Importing it here keeps a model run free of a + # plotting stack it never uses -- which is the property that made moving these + # methods off `Catchment` worth doing rather than just tidier. + from cleopatra.glyphs.gridded.array_glyph import ArrayGlyph, PointOverlay + + if option not in _ANIMATION_OPTIONS: + raise ValueError( + f"the option parameter takes a value between 1 and " + f"{max(_ANIMATION_OPTIONS)}, given: {option}" + ) + + run = self._require_distributed_run() + start_i, end_i = self._step_bounds(run.period, start, end, fmt, inclusive=False) + + source, title = _ANIMATION_OPTIONS[option] + arr = self._select(source, start_i, end_i) + + # mask the no-data cells on a copy so plotting never mutates the result arrays + arr = arr.copy() + arr[np.isnan(run.flow_network.flow_acc_arr), :] = np.nan + + time = run.period.date_index[start_i:end_i] + + if gauges is not None: + # animate expects a 3-column array: [value to display, cell row, cell column]. + # cleopatra 0.30 stopped accepting a bare array; it must be wrapped in a + # PointOverlay, which also carries the marker/label styling. + kwargs["points"] = PointOverlay( + gauges[["id", "cell_row", "cell_col"]].to_numpy() + ) + + # animate iterates over the first dimension, so move the time axis to the front + array = ArrayGlyph(np.moveaxis(arr, -1, 0)) + # the option title is a default; an explicit title= kwarg wins + kwargs.setdefault("title", title) + # cleopatra is untyped, so name what it hands back rather than letting `Any` leak + # out of a public signature. + anim: matplotlib.animation.FuncAnimation = array.animate(time, **kwargs) + + self._animation_glyph = array + self.anim = anim + + return anim + + def save_animation(self, path: str, fps: int = 2) -> None: + """Save the animation built by :meth:`animate`. + + The output format is determined by the file extension. GIF uses PillowWriter; + mov/avi/mp4 require FFmpeg to be installed. + + Args: + path: Output file path. The extension determines the format (gif, mov, avi, mp4). + fps: Frames per second. Default is 2. + + Raises: + ValueError: :meth:`animate` has not been called yet, or the file format is not + supported. + FileNotFoundError: A video format is requested but FFmpeg is not installed. + """ + if self._animation_glyph is None: + raise ValueError("There is no animation to save, call `animate` first") + self._animation_glyph.save_animation(path, fps=fps) + + def save( + self, + path: str = "", + result: int = 1, + start: str | dt.datetime = "", + end: str | dt.datetime = "", + prefix: str = "", + fmt: str = "%Y-%m-%d", + flow_acc_path: str = "", + ) -> None: + """Write the results to disk: one raster per step, or a CSV for a lumped run. + + Which of the two happens is read off :attr:`routing` rather than passed in -- a + lumped run has no grid to write rasters on, and that is a property of the results. + + Args: + path: Output directory for a distributed run (created if it does not exist), or + the CSV file itself for a lumped one. Default is "", the working directory. + result: What to write. Distributed: 1 - Total discharge, 2 - Upper zone + discharge, 3 - Lower zone discharge, 4 - Snow pack, 5 - Soil moisture, + 6 - Upper zone, 7 - Lower zone, 8 - Water content. Lumped: 1 - simulated + discharge, 2 - upper zone, 3 - lower zone, 4 - the five states, 5 - all of + them. Default is 1. + start: Start of the output period. A string is parsed with `fmt`. If empty, the + run's first step. + end: End of the output period, inclusive. If empty, the run's last step. + prefix: Prefix for the raster file names. Default is "Result_". + fmt: Date format `start` and `end` are parsed with. Default is "%Y-%m-%d". + flow_acc_path: The flow-accumulation raster, used as the georeferencing template + for the written rasters. Required for a distributed run: `FlowNetwork` keeps + the accumulation *array* but not its projection, so the grid has to be read + back from the file. + + Raises: + TypeError: `path` is not a string. `outputs.results_dir` is optional in a run + configuration, so a caller forwarding it can hold None. + ValueError: `result` is not a valid option, `flow_acc_path` is missing on a + distributed run, or the results carry no run to date them by. + """ + if not isinstance(path, str): + raise TypeError( + f"path must be a string naming a directory (distributed) or a file " + f"(lumped), got {type(path).__name__}" + ) + + run = self._require_run() + start_i, end_i = self._step_bounds(run.period, start, end, fmt, inclusive=True) + + if self.routing is RoutingKind.LUMPED: + self._save_csv(run.period, path, result, start_i, end_i) + else: + self._save_rasters( + run.period, path, result, start_i, end_i, prefix, flow_acc_path + ) + + logger.debug("Data is saved successfully") + + def _save_rasters( + self, + period: SimulationPeriod, + path: str, + result: int, + start_i: int, + end_i: int, + prefix: str, + flow_acc_path: str, + ) -> None: + """Write one GeoTIFF per step off the flow-accumulation raster's grid. + + Args: + period: The run's calendar, which names the files. + path: Destination directory, created if it does not exist. + result: Which array to write. See :meth:`save`. + start_i: First step. + end_i: One past the last step. + prefix: File-name prefix. + flow_acc_path: The georeferencing template. + + Raises: + ValueError: `flow_acc_path` is empty, or `result` is not between 1 and 8. + """ + if flow_acc_path == "": + raise ValueError( + "writing rasters needs a georeferencing template; pass flow_acc_path, the " + "flow-accumulation raster the model was built on" + ) + if result not in _RASTER_OPTIONS: + raise ValueError( + f" The result parameter takes a value between 1 and " + f"{max(_RASTER_OPTIONS)}, given: {result}" + ) + + arr = self._select(_RASTER_OPTIONS[result], start_i, end_i) + + src = Dataset.read_file(flow_acc_path) + + if prefix == "": + prefix = "Result_" + + # `path` names a directory here, unlike the CSV branch where it is the file itself. + # Joined rather than concatenated: the old `path + prefix` wrote + # `some/dirResult_2009-01-01.tif` for any directory given without a trailing + # separator, which is how a directory is normally written. + if path and not os.path.isdir(path): + os.makedirs(path, exist_ok=True) + names = [ + os.path.join(path, f"{prefix}{str(i)[:10]}.tif") + for i in period.date_index[start_i:end_i] + ] + + # from_dataset is pyramids' named constructor for an in-memory scaffold off a + # template raster; the bare Datacube(src, time_length=) form it replaced is kept + # only as a legacy fallback upstream. + cube = Datacube.from_dataset(src, arr.shape[2]) + cube.values = np.moveaxis(arr, -1, 0) + cube.to_file(names) + + def _save_csv( + self, + period: SimulationPeriod, + path: str, + result: int, + start_i: int, + end_i: int, + ) -> None: + """Write a lumped run's series to a CSV. + + Args: + period: The run's calendar, which indexes the frame. + path: The CSV file to write. + result: Which series to write. See :meth:`save`. + start_i: First step. + end_i: One past the last step. + + Raises: + ValueError: `result` is not between 1 and 5. + """ + if result not in (1, 2, 3, 4, 5): + raise ValueError( + f"in lumped mode the result parameter takes a value between 1 and 5, " + f"given: {result}" + ) + + # The run's own calendar, not a fresh daily `date_range`: the old branch hard-coded + # `freq="D"`, so an hourly lumped run wrote a daily index against hourly values. + data = pd.DataFrame(index=period.date_index[start_i:end_i]) + data["date"] = ["'" + str(i)[:10] + "'" for i in data.index] + + if result in (1, 5): + # For a lumped run the total discharge *is* `Qsim`; `Run.run_lumped` only wraps + # this same array in a frame to put on the model. + data["Qsim"] = self._require_field("q_total")[start_i:end_i] + if result == 2 or result == 5: + data["Quz"] = self.quz[start_i:end_i] + if result == 3 or result == 5: + data["Qlz"] = self.qlz[start_i:end_i] + if result in (4, 5): + data[STATE_VARIABLES] = self._require_state_variables()[start_i:end_i, :] + + data.to_csv(path, index=False, float_format="%.3f") diff --git a/src/hapi/rrm/distrrm.py b/src/hapi/rrm/distrrm.py index 43857196..5c695cef 100644 --- a/src/hapi/rrm/distrrm.py +++ b/src/hapi/rrm/distrrm.py @@ -60,6 +60,10 @@ def run_lumped_model(run: DistributedRun) -> SimulationResults: if run.keep_state_variables else None ), + # Carried as provenance: the arrays are meaningless without the calendar to + # index them by and the grid to mask them with, and `SimulationResults` renders + # and writes itself. + run=run, ) states = results.state_variables diff --git a/src/hapi/run.py b/src/hapi/run.py index f0c8434c..baea5743 100644 --- a/src/hapi/run.py +++ b/src/hapi/run.py @@ -286,12 +286,12 @@ def run_maxbas(model: CatchmentLike) -> SimulationResults: outlet, summed over every cell. - `quz`: 3D array of distributed discharge for each cell. - `q_total`, `quz_routed`, `qlz_translated`: 3D per-cell fields - read by `save_results` and `plot_distributed_results`. MAXBAS + read by `results.save` and `results.animate`. MAXBAS routes each cell straight to the outlet, so a cell of `q_total` is that cell's *contribution* to the outlet — `np.nansum` over the - domain reproduces `qout`. Use - `extract_discharge` reads the routing off the results and takes the - basin-wide sum on this path automatically. + domain reproduces `qout`. `extract_discharge` reads the routing + off the results and takes the basin-wide sum on this path + automatically. Raises: ValueError: If input data arrays have inconsistent diff --git a/src/hapi/runs.py b/src/hapi/runs.py index 61c09bee..313f27ad 100644 --- a/src/hapi/runs.py +++ b/src/hapi/runs.py @@ -81,7 +81,7 @@ class DistributedRun: :meth:`~hapi.rrm.distrrm.DistributedRRM.route_maxbas_by_path_length`. keep_state_variables: Whether to allocate the per-cell state array. It is `(rows, cols, time, 5)` -- as much memory as every other result field combined -- - and nothing but `save_results` and `plot_distributed_results` reads it, so a run + and nothing but `results.save` and `results.animate` reads it, so a run that will not look at it can halve its peak allocation. Defaults to True, which is what every existing caller got; `Calibration` turns it off, because it runs the model once per trial vector and never reads the states. @@ -183,7 +183,7 @@ def from_model( with_river_geometry: Carry the river geometry through, for the flood path. skip_hydraulic_cells: Leave the river cells to a hydraulic model. keep_state_variables: Allocate the per-cell state array. False halves the run's - peak memory at the cost of `save_results` / `plot_distributed_results` + peak memory at the cost of `results.save` / `results.animate` options 4 to 8. Returns: diff --git a/src/hapi/wrapper.py b/src/hapi/wrapper.py index eeaa3f3f..d9697b5d 100644 --- a/src/hapi/wrapper.py +++ b/src/hapi/wrapper.py @@ -217,7 +217,7 @@ def run_muskingum_with_lake(run: DistributedRun, Lake: Lake) -> SimulationResult def _set_maxbas_output_fields(results: SimulationResults) -> None: """Fill the distributed output fields after a triangular (MAXBAS) run. - `save_results` and `plot_distributed_results` read `q_total`, + `results.save` and `results.animate` read `q_total`, `quz_routed` and `qlz_translated` for their discharge options. Only :meth:`DistRRM.route_muskingum` (the Muskingum path) used to set them, so after a MAXBAS run they stayed `None` and every discharge option raised @@ -261,7 +261,7 @@ def run_maxbas(run: DistributedRun) -> SimulationResults: Also fills the per-cell output fields (`q_total`, `quz_routed`, `qlz_translated`) via :meth:`_set_maxbas_output_fields`, so the - discharge options of `save_results` / `plot_distributed_results` + discharge options of `results.save` / `results.animate` work on this path; see that method for the MAXBAS semantics. Args: @@ -464,6 +464,7 @@ def run_lumped( quz=quz * factor, qlz=qlz * factor, state_variables=state_variables, + run=run, ) # The lumped total discharge is exactly what `q_total` means, so it goes there rather # than onto the catchment as `Qsim`. `Run.run_lumped` is what indexes it by the period diff --git a/tests/calibration/lumped_calibration.py b/tests/calibration/lumped_calibration.py index c4ca5c4b..c3b01318 100644 --- a/tests/calibration/lumped_calibration.py +++ b/tests/calibration/lumped_calibration.py @@ -131,4 +131,4 @@ EndDate = "2010-04-20" Path = Path + "Results-Lumped-Model" + str(dt.datetime.now())[0:10] + ".txt" -Coello.model.save_results(result=5, StartDate=StartDate, EndDate=EndDate, path=Path) +Coello.model.results.save(result=5, start=StartDate, end=EndDate, path=Path) diff --git a/tests/rrm/catchment/test_config.py b/tests/rrm/catchment/test_config.py index d8a7fe7f..228c26a2 100644 --- a/tests/rrm/catchment/test_config.py +++ b/tests/rrm/catchment/test_config.py @@ -1656,7 +1656,7 @@ def test_the_configuration_stays_reachable_on_the_model( "somewhere/else" ), f"outputs did not survive: {model.config.outputs}" assert model.config.flow_network.flow_accumulation is not None, ( - "the flow-accumulation path should stay reachable for save_results" + "the flow-accumulation path should stay reachable for results.save" ) def test_a_hand_built_model_has_no_configuration( diff --git a/tests/rrm/catchment/test_e2e_coello_from_netcdf.py b/tests/rrm/catchment/test_e2e_coello_from_netcdf.py index f506602a..60ce6f6f 100644 --- a/tests/rrm/catchment/test_e2e_coello_from_netcdf.py +++ b/tests/rrm/catchment/test_e2e_coello_from_netcdf.py @@ -216,7 +216,7 @@ def test_routing_fills_the_distributed_fields(self, muskingum_run: Catchment): Test scenario: `q_total`, `quz_routed` and `qlz_translated` back every downstream reader -- - `extract_discharge`, `save_results`, the animations. All three must come back at + `extract_discharge`, `results.save`, the animations. All three must come back at `(rows, cols, simulation_steps)` and finite inside the catchment. """ model = muskingum_run @@ -273,7 +273,7 @@ def test_saved_rasters_carry_the_routed_discharge( Test scenario: The last link, and the one nothing else exercises for the NetCDF-driven path: - `save_results` writes one raster per step, georeferenced from the flow + `results.save` writes one raster per step, georeferenced from the flow accumulation grid. Reading the first one back and comparing it to `q_total`'s first slice proves the file holds the run's own numbers rather than an empty grid. """ @@ -281,10 +281,10 @@ def test_saved_rasters_carry_the_routed_discharge( out = tmp_path / "muskingum" out.mkdir() - model.save_results(flow_acc_path=coello_acc_path, result=1, path=f"{out}/") + model.results.save(flow_acc_path=coello_acc_path, result=1, path=f"{out}/") written = sorted(out.glob("*.tif")) - assert written, "save_results must write at least one raster" + assert written, "results.save must write at least one raster" assert len(written) == len(model.period.date_index), ( f"expected one raster per step ({len(model.period.date_index)}), got {len(written)}" ) diff --git a/tests/rrm/catchment/test_fw1_output_fields.py b/tests/rrm/catchment/test_fw1_output_fields.py index b9785e15..8676dcd5 100644 --- a/tests/rrm/catchment/test_fw1_output_fields.py +++ b/tests/rrm/catchment/test_fw1_output_fields.py @@ -2,7 +2,7 @@ Only ``DistRRM.route_muskingum`` (the Muskingum path) used to set ``q_total`` / ``quz_routed`` / ``qlz_translated``, so after ``Run.run_maxbas`` they stayed ``None`` and every -discharge option of ``save_results`` / ``plot_distributed_results`` raised +discharge option of ``SimulationResults.save`` / ``.animate`` raised ``TypeError: 'NoneType' object is not subscriptable``. ``Wrapper._set_maxbas_output_fields`` now fills them; these tests pin both the values and the MAXBAS-specific semantics. """ @@ -116,8 +116,8 @@ def test_fw1_sets_the_per_cell_output_fields(coello_fw1: Catchment): coello_fw1: Coello catchment with a completed MAXBAS run. Test scenario: - These three fields back the discharge options of `save_results` and - `plot_distributed_results`. Before the fix only the Muskingum path set + These three fields back the discharge options of `results.save` and + `results.animate`. Before the fix only the Muskingum path set them, so they were `None` here and every discharge option raised. """ shape = coello_fw1.results.quz.shape @@ -214,7 +214,7 @@ def test_extract_discharge_takes_the_basin_wide_sum_after_fw1(coello_fw1: Catchm ) -def test_save_results_distributed_discharge_after_fw1( +def test_save_rasters_of_discharge_after_fw1( coello_fw1: Catchment, coello_acc_path: str, tmp_path ): """Test that the discharge results can now be written as rasters after run_maxbas. @@ -231,7 +231,7 @@ def test_save_results_distributed_discharge_after_fw1( """ out = tmp_path / "q" out.mkdir() - coello_fw1.save_results( + coello_fw1.results.save( flow_acc_path=coello_acc_path, result=1, start="2009-01-01", @@ -268,7 +268,5 @@ def test_plot_discharge_options_after_fw1(coello_fw1: Catchment, option: int): """ import matplotlib.animation - anim = coello_fw1.plot_distributed_results( - "2009-01-01", "2009-01-05", option=option - ) + anim = coello_fw1.results.animate("2009-01-01", "2009-01-05", option=option) assert isinstance(anim, matplotlib.animation.FuncAnimation) diff --git a/tests/rrm/catchment/test_plot_animation.py b/tests/rrm/catchment/test_plot_animation.py index dac92bb5..ac1631dd 100644 --- a/tests/rrm/catchment/test_plot_animation.py +++ b/tests/rrm/catchment/test_plot_animation.py @@ -1,10 +1,11 @@ -"""Smoke tests for the cleopatra-backed animation surface of Catchment.""" +"""Smoke tests for the cleopatra-backed animation surface of SimulationResults.""" import numpy as np import pytest from hapi.catchment import Catchment from hapi.inputs import FlowNetwork, MeteoInputs +from hapi.results import RoutingKind, SimulationResults from hapi.rrm.hbv_bergestrom92 import HBVBergestrom92 as HBVLumped from hapi.run import Run @@ -48,14 +49,25 @@ def coello_animated( return coello +@pytest.fixture +def bare_results() -> SimulationResults: + """Result arrays built by hand: no run behind them and no animation yet.""" + cube = np.zeros((2, 3, 4), dtype="float32") + return SimulationResults(RoutingKind.MUSKINGUM, cube, cube, None) + + @pytest.mark.plot def test_plot_precipitation_with_gauges(coello_animated: Catchment): """Animating a meteo input with gauge points returns a FuncAnimation.""" import matplotlib.animation before = coello_animated.meteo.precipitation.copy() - anim = coello_animated.plot_distributed_results( - "2009-01-01", "2009-01-09", option=9, gauges=True, interval=100 + anim = coello_animated.results.animate( + "2009-01-01", + "2009-01-09", + option=9, + gauges=coello_animated.GaugesTable, + interval=100, ) assert isinstance(anim, matplotlib.animation.FuncAnimation) # plotting must not mutate the model arrays stored on the instance @@ -68,9 +80,7 @@ def test_plot_state_variable(coello_animated: Catchment): import matplotlib.animation before = coello_animated.results.state_variables.copy() - anim = coello_animated.plot_distributed_results( - "2009-01-01", "2009-01-09", option=5 - ) + anim = coello_animated.results.animate("2009-01-01", "2009-01-09", option=5) assert isinstance(anim, matplotlib.animation.FuncAnimation) assert np.array_equal( before, coello_animated.results.state_variables, equal_nan=True @@ -80,7 +90,7 @@ def test_plot_state_variable(coello_animated: Catchment): @pytest.mark.plot def test_plot_title_override(coello_animated: Catchment): """An explicit title= kwarg overrides the option default.""" - anim = coello_animated.plot_distributed_results( + anim = coello_animated.results.animate( "2009-01-01", "2009-01-09", option=9, title="Custom" ) assert anim is not None @@ -94,7 +104,7 @@ def test_plot_accepts_grouped_style_objects(coello_animated: Catchment): cleopatra 0.30 replaced the loose styling keywords (`color_scale`, `display_cell_value`, `num_size`, `background_color_threshold`, `text_loc`) with typed group objects, and a removed keyword now raises - rather than being silently ignored. `plot_distributed_results` forwards + rather than being silently ignored. `animate` forwards `**kwargs` untouched, so this pins that the group objects pass through — and that Hapi never re-introduces a loose keyword that would raise. """ @@ -103,11 +113,11 @@ def test_plot_accepts_grouped_style_objects(coello_animated: Catchment): from cleopatra.styling.params import CellValues from cleopatra.styling.scaling import ColorScaling - anim = coello_animated.plot_distributed_results( + anim = coello_animated.results.animate( "2009-01-01", "2009-01-09", option=9, - gauges=True, + gauges=coello_animated.GaugesTable, interval=100, color=ColorScaling.power(gamma=0.5), cells=CellValues(show=True, size=8, background_threshold=None), @@ -143,13 +153,13 @@ def spy(self, time, *args, **kwargs): return original(self, time, *args, **kwargs) monkeypatch.setattr(ArrayGlyph, "animate", spy) - coello_animated.plot_distributed_results( - "2009-01-01", "2009-01-09", option=9, gauges=True + coello_animated.results.animate( + "2009-01-01", "2009-01-09", option=9, gauges=coello_animated.GaugesTable ) points = seen["points"] assert isinstance(points, PointOverlay), ( - "gauges=True must pass a PointOverlay; a bare array raises on cleopatra >=0.30" + "a gauge table must pass a PointOverlay; a bare array raises on cleopatra >=0.30" ) assert points.points.shape[1] == 3, "points must stay [value, row, col]" @@ -157,19 +167,22 @@ def spy(self, time, *args, **kwargs): @pytest.mark.plot def test_save_animation_gif(coello_animated: Catchment, tmp_path): """save_animation writes a non-empty gif after plotting.""" - coello_animated.plot_distributed_results("2009-01-01", "2009-01-09", option=9) + coello_animated.results.animate("2009-01-01", "2009-01-09", option=9) out = tmp_path / "anim.gif" - coello_animated.save_animation(str(out), fps=2) + coello_animated.results.save_animation(str(out), fps=2) assert out.exists() assert out.stat().st_size > 0 @pytest.mark.plot -def test_save_animation_before_plot_raises(): - """save_animation without a prior plot raises a clear error.""" - coello = Catchment("bare", "2009-01-01", "2009-01-10") - with pytest.raises(ValueError, match="plot_distributed_results"): - coello.save_animation("never.gif") +def test_save_animation_before_plot_raises(bare_results: SimulationResults): + """save_animation without a prior animate raises a clear error. + + Args: + bare_results: Result arrays with no run and no animation behind them. + """ + with pytest.raises(ValueError, match="call `animate` first"): + bare_results.save_animation("never.gif") @pytest.mark.plot @@ -210,7 +223,7 @@ def spy(self, time, *args, **kwargs): monkeypatch.setattr(ArrayGlyph, "animate", spy) - coello_animated.plot_distributed_results("2009-01-01", "2009-01-09", option=option) + coello_animated.results.animate("2009-01-01", "2009-01-09", option=option) assert seen["kwargs"].get("title") == title, ( f"option {option} should be titled {title!r}, got {seen['kwargs'].get('title')!r}" @@ -222,18 +235,36 @@ def spy(self, time, *args, **kwargs): @pytest.mark.plot @pytest.mark.parametrize("option", [0, 12]) -def test_plot_invalid_option_raises(option: int): +def test_plot_invalid_option_raises(bare_results: SimulationResults, option: int): """An option outside 1-11 raises ValueError before touching any array. Args: + bare_results: Result arrays with no run behind them. option: An out-of-range plotting option. Test scenario: - The option dispatch must reject values outside 1..11 with a clear - error even on a catchment with no data loaded. + The option dispatch must reject values outside 1..11 with a clear error before it + reaches for the run, the calendar or any array -- so an out-of-range option is + reported as such rather than as whatever happens to be missing first. """ - coello = Catchment( - "bare", "2009-01-01", "2009-01-10", spatial_resolution="Distributed" - ) - with pytest.raises(ValueError, match="1 to 11"): - coello.plot_distributed_results("2009-01-01", "2009-01-09", option=option) + with pytest.raises(ValueError, match="between 1 and 11"): + bare_results.animate("2009-01-01", "2009-01-09", option=option) + + +@pytest.mark.plot +def test_results_without_a_run_say_so(bare_results: SimulationResults): + """Test that arrays with no run behind them name what is missing. + + Args: + bare_results: Result arrays built directly rather than by a run. + + Test scenario: + `animate` and `save` need the calendar to index the arrays by and the grid to mask + them with, both of which live on the run. A results object built by hand carries + neither, and used to fail on `None` several frames in. + """ + with pytest.raises(ValueError, match="carry no run"): + bare_results.animate("2009-01-01", "2009-01-09", option=1) + + with pytest.raises(ValueError, match="carry no run"): + bare_results.save(path="never") diff --git a/tests/rrm/catchment/test_rrm_catchment.py b/tests/rrm/catchment/test_rrm_catchment.py index 3b6cc7dd..ff9c8bc7 100644 --- a/tests/rrm/catchment/test_rrm_catchment.py +++ b/tests/rrm/catchment/test_rrm_catchment.py @@ -109,7 +109,7 @@ def test_save_lumped_results( coello.read_discharge_gauges(lumped_gauges_path, fmt=coello_gauges_date_fmt) Route = 1 Run.run_lumped(coello, Route, Routing.muskingum_v) - coello.save_results(result=5, path=path) + coello.results.save(result=5, path=path) # # TODO: still not finished as it does not run the plotHydrograph method # def test_PlotHydrograph( @@ -399,7 +399,7 @@ def test_extract_results( class TestSaveAndExtractAfterFW1: - """A second FW1 run covering `save_results` and `extract_discharge`. + """A second FW1 run covering `results.save` and `extract_discharge`. Named for what it does: it reads the MAXBAS parameter set and calls `Run.run_maxbas`, so calling it `TestMuskingum` said the opposite of what it exercises. The Muskingum path diff --git a/tests/rrm/catchment/test_run_narrowing.py b/tests/rrm/catchment/test_run_narrowing.py index 3db2cc1c..f2e83091 100644 --- a/tests/rrm/catchment/test_run_narrowing.py +++ b/tests/rrm/catchment/test_run_narrowing.py @@ -261,7 +261,7 @@ def test_dropping_them_halves_the_result_arrays(self, built: Catchment): Test scenario: `state_variables` is `(rows, cols, time, 5)` -- as much memory as every other - result field combined -- and only `save_results` and `plot_distributed_results` + result field combined -- and only `results.save` and `results.animate` read it. A run that will not look at it should not pay for it. """ kept = Wrapper.run_muskingum(DistributedRun.from_model(built)) @@ -325,7 +325,7 @@ def test_a_state_option_says_why_it_cannot_be_saved(self, built: Catchment): ) with pytest.raises(ValueError, match="keep_state_variables"): - built.plot_distributed_results("2009-01-01", "2009-01-05", option=4) + built.results.animate("2009-01-01", "2009-01-05", option=4) def test_a_discharge_option_still_works_without_them(self, built: Catchment): """Test that the guard only fires for the options that need the states. diff --git a/tests/rrm/catchment/test_run_results_coupling.py b/tests/rrm/catchment/test_run_results_coupling.py index 5df8d8cc..a5af90a0 100644 --- a/tests/rrm/catchment/test_run_results_coupling.py +++ b/tests/rrm/catchment/test_run_results_coupling.py @@ -173,6 +173,35 @@ def test_run_does_not_import_catchment_at_runtime(self): "hapi.protocols exist so the run layer does not depend on the concrete class" ) + def test_running_a_model_does_not_import_a_plotting_stack(self): + """Test that the run layer stays free of matplotlib and cleopatra. + + Test scenario: + `SimulationResults` renders and writes itself -- `animate`, `save_animation` and + `save` moved there off `Catchment`. Since `hapi.results` is what every engine + imports, a module-scope cleopatra import there would put a plotting stack in the + path of every model run, which is worse than the arrangement it replaced. The + import is inside `animate` for exactly that reason, and this is what holds it + there. + """ + probe = ( + "import sys; import hapi.run, hapi.wrapper; from hapi.rrm import distrrm; " + "print(','.join(m for m in ('cleopatra', 'matplotlib') if m in sys.modules) " + "or 'clean')" + ) + completed = subprocess.run( + [sys.executable, "-c", probe], + capture_output=True, + text=True, + check=True, + ) + + assert completed.stdout.strip() == "clean", ( + f"running a model must not import a plotting stack, got " + f"{completed.stdout.strip()}; keep the cleopatra import inside " + f"SimulationResults.animate" + ) + def test_run_offers_nothing_a_configuration_could_build(self): """Test that `Run` exposes no constructor-like surface. diff --git a/tests/rrm/catchment/test_save_results_distributed.py b/tests/rrm/catchment/test_save_results_distributed.py index e025052f..46a90c4f 100644 --- a/tests/rrm/catchment/test_save_results_distributed.py +++ b/tests/rrm/catchment/test_save_results_distributed.py @@ -1,4 +1,4 @@ -"""Tests for the distributed branch of ``Catchment.save_results``. +"""Tests for the raster branch of ``SimulationResults.save``. The distributed branch scaffolds an in-memory ``DatasetCollection`` off the flow-accumulation raster and writes one GeoTIFF per timestep. It was previously exercised only by the @@ -55,7 +55,7 @@ def coello_run( return coello -def test_save_results_distributed_writes_one_raster_per_step( +def test_save_writes_one_raster_per_step( coello_run: Catchment, coello_acc_path: str, tmp_path ): """Test that the distributed branch writes one readable GeoTIFF per timestep. @@ -66,7 +66,7 @@ def test_save_results_distributed_writes_one_raster_per_step( tmp_path: Destination directory. Test scenario: - `save_results` scaffolds a `DatasetCollection` from the flow-accumulation + `save` scaffolds a `DatasetCollection` from the flow-accumulation raster via pyramids' `from_dataset` named constructor and writes the selected result to one raster per date. Pins that the files land, are readable, and carry the template's grid. @@ -76,7 +76,7 @@ def test_save_results_distributed_writes_one_raster_per_step( # result=4 is the snow-pack state variable. The discharge options are available after # a FW1 run too since `_set_maxbas_output_fields` landed; this covers the state-variable # branch, which reads a different array. - coello_run.save_results( + coello_run.results.save( flow_acc_path=coello_acc_path, result=4, start="2009-01-01", @@ -94,7 +94,7 @@ def test_save_results_distributed_writes_one_raster_per_step( ) -def test_save_results_distributed_values_match_the_model_array( +def test_save_values_match_the_model_array( coello_run: Catchment, coello_acc_path: str, tmp_path ): """Test that the written rasters carry the model's discharge values. @@ -112,7 +112,7 @@ def test_save_results_distributed_values_match_the_model_array( """ out = tmp_path / "dist" out.mkdir() - coello_run.save_results( + coello_run.results.save( flow_acc_path=coello_acc_path, result=4, start="2009-01-01", @@ -132,7 +132,7 @@ def test_save_results_distributed_values_match_the_model_array( ) -def test_save_results_joins_a_directory_written_without_a_separator( +def test_save_joins_a_directory_written_without_a_separator( coello_run: Catchment, coello_acc_path: str, tmp_path ): """Test that a directory given without a trailing separator still writes inside it. @@ -151,7 +151,7 @@ def test_save_results_joins_a_directory_written_without_a_separator( out = tmp_path / "no-separator" out.mkdir() - coello_run.save_results( + coello_run.results.save( flow_acc_path=coello_acc_path, result=4, start="2009-01-01", @@ -164,7 +164,7 @@ def test_save_results_joins_a_directory_written_without_a_separator( ) -def test_save_results_creates_the_directory_it_is_given( +def test_save_creates_the_directory_it_is_given( coello_run: Catchment, coello_acc_path: str, tmp_path ): """Test that a destination directory that does not exist yet is created. @@ -181,7 +181,7 @@ def test_save_results_creates_the_directory_it_is_given( """ out = tmp_path / "nested" / "results" - coello_run.save_results( + coello_run.results.save( flow_acc_path=coello_acc_path, result=4, start="2009-01-01", @@ -194,7 +194,7 @@ def test_save_results_creates_the_directory_it_is_given( ) -def test_save_results_refuses_a_path_that_is_not_a_string(coello_run: Catchment): +def test_save_refuses_a_path_that_is_not_a_string(coello_run: Catchment): """Test that a non-string `path` is refused by name rather than by concatenation. Args: @@ -206,7 +206,7 @@ def test_save_results_refuses_a_path_that_is_not_a_string(coello_run: Catchment) concatenation, naming neither the argument nor what it should be. """ with pytest.raises(TypeError, match="path must be a string") as exc: - coello_run.save_results(flow_acc_path="unused", result=1, path=None) + coello_run.results.save(flow_acc_path="unused", result=1, path=None) assert "NoneType" in str(exc.value), ( f"the error should name what it got: {exc.value}" diff --git a/tests/rrm/catchment/test_wrapper_with_lake.py b/tests/rrm/catchment/test_wrapper_with_lake.py index e17997a6..b57b58a0 100644 --- a/tests/rrm/catchment/test_wrapper_with_lake.py +++ b/tests/rrm/catchment/test_wrapper_with_lake.py @@ -417,7 +417,7 @@ def test_fills_the_distributed_output_fields( """Test that the triangular lake path populates `q_total`, `quz_routed`, `qlz_translated`. Test scenario: - `save_results` and `plot_distributed_results` read those three fields, and only + `results.save` and `results.animate` read those three fields, and only the Muskingum path used to set them. The fixture is function-scoped so the model has been through `run_maxbas_with_lake` and nothing else -- with a shared instance an earlier Muskingum run would have filled them and deleting the fix left this green. diff --git a/tests/run/distributed_mode_run.py b/tests/run/distributed_mode_run.py index 9ccb02a8..0f2ccd94 100644 --- a/tests/run/distributed_mode_run.py +++ b/tests/run/distributed_mode_run.py @@ -82,7 +82,7 @@ """ Animate the distributed results. -plot_distributed_results forwards the keyword arguments to +SimulationResults.animate forwards the keyword arguments to ``cleopatra.glyphs.gridded.array_glyph.ArrayGlyph.animate``; see its docstring for the full list of supported options. Since cleopatra 0.30 the styling keywords are grouped into typed objects (``color``, ``cells``, ``frame_label``). @@ -91,7 +91,7 @@ plotstart = "2009-01-01" plotend = "2009-02-01" -Anim = Coello.plot_distributed_results( +Anim = Coello.results.animate( plotstart, plotend, figsize=(9, 9), @@ -99,7 +99,7 @@ cells=CellValues(show=True, background_threshold=160), ticks_spacing=5, interval=200, - gauges=True, + gauges=Coello.GaugesTable, cmap="inferno", frame_label=FrameLabel(location=[0.1, 0.2]), color=ColorScaling.linear(), @@ -107,18 +107,18 @@ # %% SaveTo = Path + "/results/anim.gif" -Coello.save_animation(SaveTo, fps=2) +Coello.results.save_animation(SaveTo, fps=2) # %% Save the result into rasters StartDate = "2009-01-01" EndDate = "2009-04-10" Prefix = "Qtot_" SaveTo = Path + "/results/" -Coello.save_results( - FlowAccPath, - result=1, - StartDate=StartDate, - EndDate=EndDate, +Coello.results.save( path=SaveTo, + flow_acc_path=FlowAccPath, + result=1, + start=StartDate, + end=EndDate, prefix=Prefix, ) diff --git a/tests/run/lumped_run.py b/tests/run/lumped_run.py index a09835bb..7431bf14 100644 --- a/tests/run/lumped_run.py +++ b/tests/run/lumped_run.py @@ -65,4 +65,4 @@ EndDate = "2010-04-20" Path = Path + "Results-Lumped-Model" + str(dt.datetime.now())[0:10] + ".txt" -Coello.save_results(result=1, StartDate=StartDate, EndDate=EndDate, path=Path) +Coello.results.save(result=1, start=StartDate, end=EndDate, path=Path) From b6cdbbb18da25d7358e25c7c67b9495060071b94 Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Thu, 10 Sep 2026 21:36:55 +0200 Subject: [PATCH 18/54] test(results): cover the edges of the methods that moved off Catchment `hapi.results` sat at 90% line and branch coverage after the presentation methods landed on it. The happy paths were covered -- they came with the methods -- but thirteen branches were not, and each one is a place where a caller gets an answer they did not ask for rather than an error: * a lumped run asked to animate, which has no grid and no drivers; * a routed field read before its routing step filled it; * `start` / `end` given as `datetime` rather than `str`; * a date that is not a step of the run at all -- `np.nonzero(...)[0][0]` raises `IndexError: index 0 is out of bounds`, naming neither the date nor the span; * rasters asked for with no georeferencing template, or an option outside 1-8; * a caller's raster prefix, which every existing test left at the default; * the lumped CSV options 2, 3 and 4, and an option outside 1-5, which would otherwise write a file holding nothing but a `date` column. `tests/rrm/catchment/test_results.py` takes the ones that are about the object itself; the three raster guards go beside the fixture that already builds rasters in `test_save_results_distributed.py`. `hapi.results` is now at 100% line and branch, and the suite is 627 in the main task plus 17 in `plot`. --- tests/rrm/catchment/test_results.py | 449 ++++++++++++++++++ .../test_save_results_distributed.py | 78 +++ 2 files changed, 527 insertions(+) create mode 100644 tests/rrm/catchment/test_results.py diff --git a/tests/rrm/catchment/test_results.py b/tests/rrm/catchment/test_results.py new file mode 100644 index 00000000..b61ffaf1 --- /dev/null +++ b/tests/rrm/catchment/test_results.py @@ -0,0 +1,449 @@ +"""Tests for `SimulationResults`' own surface: what it refuses, and what it writes. + +The presentation methods moved here off `Catchment`, so this is where the arrays are read, +dated and written. The happy paths live beside the fixtures that produce them -- +`test_save_results_distributed.py` for the rasters, `test_plot_animation.py` for the +animation, `test_rrm_catchment.py` for the lumped CSV. What is left, and what this file +covers, is the behaviour at the edges of those methods: the run they need and might not have, +the fields a routing step has not filled yet, the dates that are not steps of the run, and the +CSV options nothing else exercises. +""" + +from __future__ import annotations + +import datetime as dt + +import numpy as np +import pandas as pd +import pytest + +from hapi.catchment import Catchment +from hapi.inputs import FlowNetwork, MeteoInputs +from hapi.results import STATE_VARIABLES, RoutingKind, SimulationResults +from hapi.routing import Routing +from hapi.rrm.distrrm import DistributedRRM +from hapi.rrm.hbv_bergestrom92 import HBVBergestrom92 as HBVLumped +from hapi.run import Run +from hapi.runs import DistributedRun + +DATE_REGEX = r"\d{4}.\d{2}.\d{2}" + + +@pytest.fixture(scope="module") +def distributed_run( + coello_start_date: str, + coello_end_date: str, + coello_prec_path: str, + coello_temp_path: str, + coello_evap_path: str, + coello_acc_path: str, + coello_fd_path: str, + coello_dist_parameters_muskingum: str, + coello_cat_area: int, + coello_initial_cond: list, +) -> DistributedRun: + """A validated distributed Coello run, with no engine having touched it yet. + + Returns: + DistributedRun: The narrowed run. + """ + model = Catchment( + "coello", + coello_start_date, + coello_end_date, + spatial_resolution="Distributed", + temporal_resolution="Daily", + ) + model.meteo = MeteoInputs.from_rasters( + coello_prec_path, + coello_temp_path, + coello_evap_path, + start=coello_start_date, + end=coello_end_date, + regex_string=DATE_REGEX, + date=True, + file_name_data_fmt="%Y.%m.%d", + ) + model.flow_network = FlowNetwork.from_rasters(coello_acc_path, coello_fd_path) + model.read_parameters(coello_dist_parameters_muskingum, False) + model.read_lumped_model(HBVLumped, coello_cat_area, coello_initial_cond) + return DistributedRun.from_model(model) + + +@pytest.fixture(scope="module") +def unrouted(distributed_run: DistributedRun) -> SimulationResults: + """Results straight out of the per-cell model, before any routing step. + + Args: + distributed_run: The validated run. + + Returns: + SimulationResults: `quz`, `qlz` and the states filled; every routed field `None`. + """ + return DistributedRRM.run_lumped_model(distributed_run) + + +@pytest.fixture(scope="module") +def lumped_results( + coello_rrm_date: list, + lumped_meteo_data_path: str, + coello_AreaCoeff: float, + coello_InitialCond: list, + lumped_parameters_path: str, + coello_Snow: int, +) -> SimulationResults: + """A finished lumped run's results. + + Returns: + SimulationResults: Routed `RoutingKind.LUMPED`, carrying its `LumpedRun`. + """ + model = Catchment("rrm", coello_rrm_date[0], coello_rrm_date[1]) + model.read_lumped_inputs(lumped_meteo_data_path) + model.read_lumped_model(HBVLumped, coello_AreaCoeff, coello_InitialCond) + model.read_parameters(lumped_parameters_path, coello_Snow) + return Run.run_lumped(model, 1, Routing.muskingum_v) + + +class TestTheRunTravelsWithTheArrays: + """`run` is what makes the arrays interpretable, so its absence is reported by name.""" + + def test_a_finished_run_is_carried_on_its_results( + self, distributed_run: DistributedRun, unrouted: SimulationResults + ): + """Test that the engine attaches the run it was given to what it returns. + + Args: + distributed_run: The validated run. + unrouted: What the per-cell model produced from it. + + Test scenario: + Without this the arrays have no calendar to be indexed by and no grid to be + masked with, and every presentation method would need them passed back in. + """ + assert unrouted.run is distributed_run, ( + "the results must carry the run that produced them, not a copy" + ) + + def test_a_lumped_run_is_carried_too(self, lumped_results: SimulationResults): + """Test that the lumped engine attaches its run as well. + + Args: + lumped_results: A finished lumped run's results. + + Test scenario: + `save` dates the CSV from `run.period`, so the lumped path needs the run just as + much as the distributed one -- it only needs less of it. + """ + assert lumped_results.run is not None, "a lumped run is provenance too" + assert lumped_results.routing is RoutingKind.LUMPED, ( + f"expected LUMPED, got {lumped_results.routing}" + ) + + def test_a_lumped_run_cannot_be_animated(self, lumped_results: SimulationResults): + """Test that asking a lumped run for an animation says why it cannot. + + Args: + lumped_results: A finished lumped run's results. + + Test scenario: + A lumped run has no grid and no spatial drivers, so there is nothing to animate. + `LumpedRun` carries neither `flow_network` nor `meteo`, so without this guard the + failure would be an `AttributeError` naming a field of the run rather than the + reason. + """ + with pytest.raises(ValueError, match="lumped run"): + lumped_results.animate("2012-06-14", "2012-06-20", option=1) + + +class TestUnfilledFieldsAreNamed: + """A routed field is `None` until its routing step runs, and that is a real state.""" + + @pytest.mark.parametrize("option, field", [(1, "q_total"), (2, "quz_routed")]) + def test_an_unrouted_field_names_itself_and_the_routing( + self, unrouted: SimulationResults, option: int, field: str + ): + """Test that reading a routed field before routing names the field and the state. + + Args: + unrouted: Results from the per-cell model, with no routing applied. + option: The animation option that reads the field. + field: The field it reads. + + Test scenario: + `run_lumped_model` leaves every routed field `None` and the routing `UNROUTED`. + Slicing one used to raise `TypeError: 'NoneType' object is not subscriptable` + from inside the option dispatch, naming neither the field nor the missing step. + """ + with pytest.raises(ValueError, match=field) as exc: + unrouted.animate("2009-01-01", "2009-01-05", option=option) + + assert "unrouted" in str(exc.value), ( + f"the error should say which state the results are in: {exc.value}" + ) + + def test_an_unread_state_option_names_the_switch( + self, distributed_run: DistributedRun + ): + """Test that a state option on a run that dropped the states names the switch. + + Args: + distributed_run: The validated run, rebuilt here without the states. + + Test scenario: + `keep_state_variables=False` is the memory opt-out; the guard has to name it + rather than fail on `None` inside a slice several frames away. + """ + model = distributed_run + dropped = DistributedRun( + period=model.period, + meteo=model.meteo, + flow_network=model.flow_network, + parameters=model.parameters, + model_setup=model.model_setup, + keep_state_variables=False, + ) + results = DistributedRRM.run_lumped_model(dropped) + + with pytest.raises(ValueError, match="keep_state_variables"): + results.animate("2009-01-01", "2009-01-05", option=4) + + +class TestDatesAreResolvedAgainstTheRun: + """The arrays are positional; the run's calendar is what turns a date into a step.""" + + def test_datetime_arguments_are_accepted_as_they_are( + self, unrouted: SimulationResults, tmp_path + ): + """Test that `datetime` bounds are used directly rather than re-parsed. + + Args: + unrouted: Results carrying the run whose calendar dates them. + tmp_path: Unused destination; the call is expected to fail before writing. + + Test scenario: + `start` and `end` are documented as `str | datetime`. A string is read with + `fmt`; a datetime must skip that, since `strptime` on one raises `TypeError`. + Reaching the *result*-option error proves both bounds resolved. + """ + with pytest.raises(ValueError, match="between 1 and 8"): + unrouted.save( + path=str(tmp_path), + result=99, + start=dt.datetime(2009, 1, 1), + end=dt.datetime(2009, 1, 5), + flow_acc_path="unused", + ) + + @pytest.mark.parametrize( + "start, end, offender", + [ + ("2008-12-31", "2009-01-05", "start"), + ("2009-01-01", "2030-01-01", "end"), + ], + ) + def test_a_date_outside_the_run_is_refused( + self, unrouted: SimulationResults, start: str, end: str, offender: str + ): + """Test that a date the run never covered is named rather than silently missed. + + Args: + unrouted: Results carrying the run whose calendar dates them. + start: Start date under test. + end: End date under test. + offender: Which of the two is out of range. + + Test scenario: + The lookup is `np.nonzero(index == value)[0][0]`, which raises + `IndexError: index 0 is out of bounds` on a miss -- naming neither the date nor + the span the run actually covers. + """ + with pytest.raises(ValueError, match=f"{offender} date") as exc: + unrouted.animate(start, end, option=1) + + assert "which covers" in str(exc.value), ( + f"the error should show the span the run does cover: {exc.value}" + ) + + def test_empty_bounds_mean_the_whole_run( + self, lumped_results: SimulationResults, tmp_path + ): + """Test that omitting both dates writes the run's whole span. + + Args: + lumped_results: A finished lumped run's results. + tmp_path: Destination directory. + + Test scenario: + `""` is the documented default for both bounds and means "the first step" and + "the last step". The row count is what proves it, since a wrong default would + still write a valid file. + """ + out = tmp_path / "whole.csv" + + lumped_results.save(path=str(out), result=1) + + written = pd.read_csv(out) + assert len(written) == len(lumped_results.run.period), ( + f"expected one row per step ({len(lumped_results.run.period)}), " + f"got {len(written)}" + ) + + +class TestTheCsvBranch: + """A lumped run has no grid, so `save` writes a CSV -- read off `routing`, not passed.""" + + @pytest.mark.parametrize( + "result, columns", + [ + (1, ["Qsim"]), + (2, ["Quz"]), + (3, ["Qlz"]), + (4, STATE_VARIABLES), + (5, ["Qsim", "Quz", "Qlz", *STATE_VARIABLES]), + ], + ) + def test_each_option_writes_its_own_columns( + self, lumped_results: SimulationResults, tmp_path, result: int, columns: list + ): + """Test that every lumped option writes exactly the series it names. + + Args: + lumped_results: A finished lumped run's results. + tmp_path: Destination directory. + result: The option under test. + columns: The columns it must produce, beside `date`. + + Test scenario: + The five options were a chain of `elif` branches each re-writing the file; they + are now additive, so option 5 is the union of the others rather than a separate + copy of them. Only 1 and 5 were exercised anywhere before this. + """ + out = tmp_path / f"result-{result}.csv" + + lumped_results.save(path=str(out), result=result) + + written = pd.read_csv(out) + assert written.columns.to_list() == ["date", *columns], ( + f"option {result} should write {['date', *columns]}, " + f"got {written.columns.to_list()}" + ) + assert len(written) == len(lumped_results.run.period), ( + f"expected one row per step, got {len(written)}" + ) + + def test_the_index_follows_the_run_not_a_daily_default( + self, lumped_results: SimulationResults, tmp_path + ): + """Test that the written dates come from the run's own calendar. + + Args: + lumped_results: A finished lumped run's results. + tmp_path: Destination directory. + + Test scenario: + This branch used to build its index with a hard-coded `freq="D"`, so an hourly + lumped run wrote a daily index against hourly values -- the dates and the numbers + described different steps. Comparing against `period.date_index` is what pins + that the calendar is the run's. + """ + out = tmp_path / "dates.csv" + + lumped_results.save(path=str(out), result=1) + + written = pd.read_csv(out) + expected = [ + f"'{str(step)[:10]}'" for step in lumped_results.run.period.date_index + ] + assert written["date"].to_list() == expected, ( + "the dates must be the run's own steps, not a fresh daily range" + ) + + @pytest.mark.parametrize("result", [0, 6]) + def test_an_option_outside_the_range_is_refused( + self, lumped_results: SimulationResults, tmp_path, result: int + ): + """Test that a lumped option outside 1-5 raises rather than writing a bare file. + + Args: + lumped_results: A finished lumped run's results. + tmp_path: Destination directory. + result: An out-of-range option. + + Test scenario: + Without the guard the frame would be built, no series added, and a file with + only a `date` column written -- a silent wrong answer rather than an error. + """ + out = tmp_path / "never.csv" + + with pytest.raises(ValueError, match="between 1 and 5"): + lumped_results.save(path=str(out), result=result) + + assert not out.exists(), "nothing should be written when the option is refused" + + +class TestResultsBuiltByHand: + """Not every `SimulationResults` came from a run, and the difference has to be visible.""" + + @pytest.fixture + def orphan(self) -> SimulationResults: + """Result arrays with no run behind them. + + Returns: + SimulationResults: Routed-looking arrays, built directly. + """ + cube = np.zeros((2, 3, 4), dtype="float32") + return SimulationResults(RoutingKind.MUSKINGUM, cube, cube, None, q_total=cube) + + def test_saving_without_a_run_says_what_is_missing(self, orphan, tmp_path): + """Test that `save` on hand-built results names the absent run. + + Args: + orphan: Results built directly rather than by a run. + tmp_path: Destination directory. + + Test scenario: + There is no calendar to date the files by, so the failure has to name the run + rather than surface as an `AttributeError` on `None.period`. + """ + with pytest.raises(ValueError, match="carry no run"): + orphan.save(path=str(tmp_path), flow_acc_path="unused") + + def test_a_non_string_path_is_refused_before_anything_else(self, orphan): + """Test that `path` is type-checked ahead of the run lookup. + + Args: + orphan: Results built directly rather than by a run. + + Test scenario: + `outputs.results_dir` is optional in a run configuration, so a caller forwarding + it can hold `None`. The check has to come first, or the message would name the + missing run instead of the argument the caller actually got wrong. + """ + with pytest.raises(TypeError, match="path must be a string") as exc: + orphan.save(path=None) + + assert "NoneType" in str(exc.value), ( + f"the error should name what it got: {exc.value}" + ) + + def test_the_animation_state_is_not_a_constructor_argument(self, orphan): + """Test that `anim` and the glyph start empty and are not settable at construction. + + Args: + orphan: Results built directly rather than by a run. + + Test scenario: + They are `field(init=False)` because they are set by `animate`, not by whoever + builds the results. Passing one would otherwise silently create a results object + claiming an animation it does not have. + """ + assert orphan.anim is None, "no animation until `animate` runs" + + with pytest.raises(TypeError): + SimulationResults( + RoutingKind.MUSKINGUM, + orphan.quz, + orphan.qlz, + None, + anim="not a constructor argument", + ) diff --git a/tests/rrm/catchment/test_save_results_distributed.py b/tests/rrm/catchment/test_save_results_distributed.py index 46a90c4f..0271699b 100644 --- a/tests/rrm/catchment/test_save_results_distributed.py +++ b/tests/rrm/catchment/test_save_results_distributed.py @@ -211,3 +211,81 @@ def test_save_refuses_a_path_that_is_not_a_string(coello_run: Catchment): assert "NoneType" in str(exc.value), ( f"the error should name what it got: {exc.value}" ) + + +def test_save_uses_the_prefix_it_is_given( + coello_run: Catchment, coello_acc_path: str, tmp_path +): + """Test that an explicit prefix names the files instead of the default. + + Args: + coello_run: Distributed Coello catchment with a completed run. + coello_acc_path: Path to the flow-accumulation raster used as the template. + tmp_path: Destination directory. + + Test scenario: + `prefix` defaults to `Result_` only when it is left empty, and every other test in + this file takes that default -- so the branch that keeps a caller's prefix was never + exercised. A run writing several variables into one directory depends on it. + """ + out = tmp_path / "prefixed" + out.mkdir() + + coello_run.results.save( + path=f"{out}/", + flow_acc_path=coello_acc_path, + result=1, + start="2009-01-01", + end="2009-01-02", + prefix="Qtot_", + ) + + written = sorted(p.name for p in out.glob("*.tif")) + assert written == ["Qtot_2009-01-01.tif", "Qtot_2009-01-02.tif"], ( + f"the files must carry the given prefix, got {written}" + ) + + +def test_save_without_a_template_raster_says_what_it_needs( + coello_run: Catchment, tmp_path +): + """Test that writing rasters with no `flow_acc_path` names the missing template. + + Args: + coello_run: Distributed Coello catchment with a completed run. + tmp_path: Destination directory. + + Test scenario: + `FlowNetwork` keeps the accumulation array but not its projection, so the grid has to + be read back from the file. Without the template the error used to be a pyramids + failure on an empty path, several frames from the argument the caller omitted. + """ + with pytest.raises(ValueError, match="flow_acc_path"): + coello_run.results.save(path=str(tmp_path), result=1) + + +@pytest.mark.parametrize("result", [0, 9]) +def test_save_refuses_a_raster_option_outside_the_range( + coello_run: Catchment, coello_acc_path: str, tmp_path, result: int +): + """Test that a distributed option outside 1-8 raises before any file is written. + + Args: + coello_run: Distributed Coello catchment with a completed run. + coello_acc_path: Path to the flow-accumulation raster used as the template. + tmp_path: Destination directory. + result: An out-of-range option. + + Test scenario: + The eight options map onto three result fields and five slices of the state array. + Anything else has no array to write, and the check has to come before the template + is opened so a typo does not leave a half-written directory. + """ + out = tmp_path / "never" + + with pytest.raises(ValueError, match="between 1 and 8"): + coello_run.results.save( + path=str(out), flow_acc_path=coello_acc_path, result=result + ) + + assert not out.exists(), "nothing should be created when the option is refused" From 641ed6762134b75eb9aa43b1f083eaca08a09d55 Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Thu, 10 Sep 2026 21:36:56 +0200 Subject: [PATCH 19/54] docs(results): document the moved methods, and fix the docstrings that drifted `mkdocs build --strict` failed on thirty griffe warnings, every one of them a docstring describing a signature that no longer exists. `Wrapper`'s five entry points still documented `Model`, `ll_temp`, `q_0` and `skip_hydraulic_cells` -- parameters removed when the engines started taking a `DistributedRun` or a `LumpedRun` -- and three of them told the reader that results are "stored directly on the Model object", which is the exact behaviour that change removed. They document `run` now, and say what they return. The other eleven were a bullet list under `Returns:` whose wrapped lines sat at six spaces where griffe wants a multiple of four. `mkdocs build --strict` is clean. `SimulationResults.animate`, `save_animation` and `save` arrived here with no `Examples:` at all. Each has one now, and they run: the doctest task covers this module, so `save`'s example builds a small `LumpedRun` and writes a real CSV, and the reader sees the actual columns and the date quoting rather than a description of them. The rendering path is not doctested -- it needs a raster grid and a plotting backend -- and stays covered by `test_plot_animation.py`. `animate` also imported cleopatra as its first statement, so a rejected option or an out-of-range date paid for loading matplotlib before being told it was wrong. The import sits below the checks now, which is what makes the first example true. 39 doctests pass, 3 skipped. --- src/hapi/results.py | 121 +++++++++++++++++++++++++++++++++++++++++--- src/hapi/run.py | 24 ++++----- src/hapi/wrapper.py | 121 ++++++++++++++------------------------------ 3 files changed, 164 insertions(+), 102 deletions(-) diff --git a/src/hapi/results.py b/src/hapi/results.py index 6cbcaedb..fb05c8ac 100644 --- a/src/hapi/results.py +++ b/src/hapi/results.py @@ -406,13 +406,41 @@ def animate( Raises: ValueError: `option` is not between 1 and 11, the results carry no distributed run, or a state option was asked for on a run that dropped the states. - """ - # cleopatra pulls in matplotlib, and this module is imported by the engines - # (`distrrm`, `wrapper`, `run`). Importing it here keeps a model run free of a - # plotting stack it never uses -- which is the property that made moving these - # methods off `Catchment` worth doing rather than just tidier. - from cleopatra.glyphs.gridded.array_glyph import ArrayGlyph, PointOverlay + Examples: + - An option outside the table is refused before any array is touched, and + before the plotting stack is imported: + ```python + >>> import numpy as np + >>> from hapi.results import RoutingKind, SimulationResults + >>> cube = np.zeros((2, 3, 4), dtype="float32") + >>> results = SimulationResults( + ... RoutingKind.MUSKINGUM, cube, cube, None, q_total=cube + ... ) + >>> results.animate("2009-01-01", "2009-01-02", option=99) + Traceback (most recent call last): + ... + ValueError: the option parameter takes a value between 1 and 11, given: 99 + + ``` + - Arrays with no run behind them have no calendar and no grid, so they say so + rather than failing on `None` several frames in: + ```python + >>> import numpy as np + >>> from hapi.results import RoutingKind, SimulationResults + >>> cube = np.zeros((2, 3, 4), dtype="float32") + >>> orphan = SimulationResults(RoutingKind.MUSKINGUM, cube, cube, None) + >>> orphan.animate("2009-01-01", "2009-01-02", option=1) + Traceback (most recent call last): + ... + ValueError: these results carry no run... + + ``` + + See Also: + save_animation: Writes the animation this builds. + save: Writes the same arrays as rasters or a CSV instead of rendering them. + """ if option not in _ANIMATION_OPTIONS: raise ValueError( f"the option parameter takes a value between 1 and " @@ -431,6 +459,14 @@ def animate( time = run.period.date_index[start_i:end_i] + # cleopatra pulls in matplotlib, and this module is imported by the engines + # (`distrrm`, `wrapper`, `run`). Importing it here keeps a model run free of a + # plotting stack it never uses -- which is the property that made moving these + # methods off `Catchment` worth doing rather than just tidier. It sits below the + # checks above so a rejected option or an out-of-range date does not pay for it + # either. + from cleopatra.glyphs.gridded.array_glyph import ArrayGlyph, PointOverlay + if gauges is not None: # animate expects a 3-column array: [value to display, cell row, cell column]. # cleopatra 0.30 stopped accepting a bare array; it must be wrapped in a @@ -466,6 +502,25 @@ def save_animation(self, path: str, fps: int = 2) -> None: ValueError: :meth:`animate` has not been called yet, or the file format is not supported. FileNotFoundError: A video format is requested but FFmpeg is not installed. + + Examples: + - There is nothing to write until :meth:`animate` has built it: + ```python + >>> import numpy as np + >>> from hapi.results import RoutingKind, SimulationResults + >>> cube = np.zeros((2, 3, 4), dtype="float32") + >>> results = SimulationResults(RoutingKind.MUSKINGUM, cube, cube, None) + >>> results.anim is None + True + >>> results.save_animation("flow.gif") + Traceback (most recent call last): + ... + ValueError: There is no animation to save, call `animate` first + + ``` + + See Also: + animate: Builds the animation this writes. """ if self._animation_glyph is None: raise ValueError("There is no animation to save, call `animate` first") @@ -509,6 +564,60 @@ def save( configuration, so a caller forwarding it can hold None. ValueError: `result` is not a valid option, `flow_acc_path` is missing on a distributed run, or the results carry no run to date them by. + + Examples: + - A lumped run writes a CSV, dated by the run's own calendar. Option 1 is the + simulated discharge, which for a lumped run is `q_total` itself: + ```python + >>> import os, tempfile + >>> from pathlib import Path + >>> import numpy as np + >>> from hapi.conceptual import ConceptualModelSetup, ParameterSet + >>> from hapi.period import SimulationPeriod + >>> from hapi.results import RoutingKind, SimulationResults + >>> from hapi.rrm.hbv_bergestrom92 import HBVBergestrom92 + >>> from hapi.runs import LumpedRun + >>> period = SimulationPeriod.parse("2009-01-01", "2009-01-03") + >>> run = LumpedRun( + ... period=period, + ... data=np.ones((len(period), 4)), + ... parameters=ParameterSet(np.ones(12), snow=False, maxbas=False), + ... model_setup=ConceptualModelSetup( + ... HBVBergestrom92, 100.0, [0.0] * 5, 1.0 + ... ), + ... ) + >>> discharge = np.array([1.5, 2.5, 3.5]) + >>> results = SimulationResults( + ... RoutingKind.LUMPED, discharge, discharge, None, + ... q_total=discharge, run=run, + ... ) + >>> path = os.path.join(tempfile.mkdtemp(), "q.csv") + >>> results.save(path=path, result=1) + >>> print(Path(path).read_text().strip()) + date,Qsim + '2009-01-01',1.500 + '2009-01-02',2.500 + '2009-01-03',3.500 + + ``` + - `path` is checked before anything else, because a run configuration's + `outputs.results_dir` is optional and a caller can forward `None`: + ```python + >>> import numpy as np + >>> from hapi.results import RoutingKind, SimulationResults + >>> cube = np.zeros((2, 3, 4), dtype="float32") + >>> results = SimulationResults(RoutingKind.MUSKINGUM, cube, cube, None) + >>> results.save(path=None) + Traceback (most recent call last): + ... + TypeError: path must be a string naming a directory (distributed) or a file (lumped), got NoneType + + ``` + + See Also: + animate: Renders the same arrays instead of writing them. + hapi.runs.DistributedRun.keep_state_variables: Whether the state options have + anything to write. """ if not isinstance(path, str): raise TypeError( diff --git a/src/hapi/run.py b/src/hapi/run.py index baea5743..b1b745a1 100644 --- a/src/hapi/run.py +++ b/src/hapi/run.py @@ -158,15 +158,16 @@ def run_distributed(model: CatchmentLike) -> SimulationResults: SimulationResults: The run's output, also assigned to `model.results`: - `state_variables`: 4D array (rows, cols, time, states) where - states are [sp, wc, sm, uz, lv]. + states are [sp, wc, sm, uz, lv]. - `qlz`: 3D array of the lower zone discharge. - `quz`: 3D array of the upper zone discharge. - `quz_routed`: 3D array of the upper zone discharge - accumulated and routed at each time step. + accumulated and routed at each time step. - `qlz_translated`: 3D array of the lower zone discharge - translated at each time step. - - `q_total`: `quz_routed + qlz_translated`. Routed by Muskingum, so the outlet - cell carries the outlet hydrograph; `extract_discharge` fills `qout` from it. + translated at each time step. + - `q_total`: `quz_routed + qlz_translated`. Routed by Muskingum, so the + outlet cell carries the outlet hydrograph; `extract_discharge` fills + `qout` from it. Raises: ValueError: If input data arrays have inconsistent @@ -283,15 +284,14 @@ def run_maxbas(model: CatchmentLike) -> SimulationResults: - `state_variables`: 4D array of state variables. - `qout`: 1D array of calculated discharge at the catchment - outlet, summed over every cell. + outlet, summed over every cell. - `quz`: 3D array of distributed discharge for each cell. - `q_total`, `quz_routed`, `qlz_translated`: 3D per-cell fields - read by `results.save` and `results.animate`. MAXBAS - routes each cell straight to the outlet, so a cell of `q_total` is - that cell's *contribution* to the outlet — `np.nansum` over the - domain reproduces `qout`. `extract_discharge` reads the routing - off the results and takes the basin-wide sum on this path - automatically. + read by `results.save` and `results.animate`. MAXBAS routes each + cell straight to the outlet, so a cell of `q_total` is that cell's + *contribution* to the outlet — `np.nansum` over the domain + reproduces `qout`. `extract_discharge` reads the routing off the + results and takes the basin-wide sum on this path automatically. Raises: ValueError: If input data arrays have inconsistent diff --git a/src/hapi/wrapper.py b/src/hapi/wrapper.py index d9697b5d..ee633f65 100644 --- a/src/hapi/wrapper.py +++ b/src/hapi/wrapper.py @@ -80,33 +80,12 @@ def run_muskingum(run: DistributedRun) -> SimulationResults: 2. The spatial routing scheme that routes flow following the river network. - The method stores results directly on the Model object, - including `quz`, `qlz`, `qout`, `quz_routed`, and - `qlz_translated` arrays. - Args: - Model: Catchment model object containing: - - - meteo (:class:`~hapi.inputs.MeteoInputs`): The three driver cubes, - each `(rows, cols, time)`, plus the calendar they cover. - - flow_network (:class:`~hapi.inputs.FlowNetwork`): The flow accumulation - and direction arrays, the direction table, and the grid they define. - - parameters (numpy.ndarray): 3D array of spatially distributed catchment - parameters, `(rows, cols, n_parameters)`. - - conversion_factor (float): Depth-to-discharge factor for the temporal - resolution; 24 for daily, 1 for hourly. - - area (float): Catchment area in km2. - - initial_cond (list): Initial state variable values - [sp, sm, uz, lz, wc]. - - snow (int): 1 to run the snow routine, 0 otherwise. - - ll_temp (numpy.ndarray, optional): 3D array of long-term - average temperature data. Defaults to None. - q_0 (float, optional): Initial discharge in m3/s. - Defaults to None. - skip_hydraulic_cells (bool, optional): Leave cells with a positive - `bankfull_depth` unrouted, because a 1D hydraulic model routes them. - The flood model's path. Defaults to False. + run: The validated inputs. Reads the drivers, the flow network, the parameter + cube and the conceptual model setup, and honours + :attr:`~hapi.runs.DistributedRun.skip_hydraulic_cells`, which leaves cells + with a positive `river_geometry.bankfull_depth` for a 1D hydraulic model + to route instead. Returns: SimulationResults: The run's output. Nothing is written to the caller's model; @@ -132,9 +111,9 @@ def run_muskingum_with_lake(run: DistributedRun, Lake: Lake) -> SimulationResult routing. Args: - Model: Catchment model object containing the distributed - model configuration, parameters, and spatial data. - Lake: Lake object containing: + run: The validated inputs. See :class:`~hapi.runs.DistributedRun`, which + `DistributedRun.from_model(model)` builds and checks. + Lake: The lake record, carrying: - MeteoData (numpy.ndarray): 2D array with columns for precipitation, evapotranspiration, temperature, @@ -148,10 +127,9 @@ def run_muskingum_with_lake(run: DistributedRun, Lake: Lake) -> SimulationResult - OutflowCell (tuple): Row and column indices of the lake outflow cell. - ll_temp (numpy.ndarray, optional): 3D array of long-term - average temperature data. Defaults to None. - q_0 (float, optional): Initial discharge in m3/s. - Defaults to None. + Returns: + SimulationResults: The run's output. Nothing is written to the caller's model; + the entry point in :mod:`hapi.run` is what puts it on `model.results`. """ meteo_data, lake_parameters, outflow_cell = _lake_inputs(Lake) plake = meteo_data[:, 0] @@ -237,8 +215,8 @@ def _set_maxbas_output_fields(results: SimulationResults) -> None: nothing downstream writes through the alias. Args: - Model: Catchment whose `quz` / `qlz` have been routed by - :meth:`DistRRM.route_maxbas`. + results: The results whose `quz` / `qlz` have been routed by + :meth:`~hapi.rrm.distrrm.DistributedRRM.route_maxbas`. Mutated in place. """ results.quz_routed = results.quz results.qlz_translated = results.qlz @@ -265,12 +243,12 @@ def run_maxbas(run: DistributedRun) -> SimulationResults: work on this path; see that method for the MAXBAS semantics. Args: - Model: Catchment model object containing the distributed - model configuration, parameters, and spatial data. - ll_temp (numpy.ndarray, optional): 3D array of long-term - average temperature data. Defaults to None. - q_0 (float, optional): Initial discharge in m3/s. - Defaults to None. + run: The validated inputs. See :class:`~hapi.runs.DistributedRun`, which + `DistributedRun.from_model(model)` builds and checks. + + Returns: + SimulationResults: The run's output. Nothing is written to the caller's model; + the entry point in :mod:`hapi.run` is what puts it on `model.results`. """ # subcatchment results = distrrm.run_lumped_model(run) @@ -305,24 +283,14 @@ def run_maxbas_with_lake(run: DistributedRun, Lake: Lake) -> SimulationResults: has been routed using the triangular function. Args: - Model: Catchment model object containing the distributed - model configuration, parameters, and spatial data. - Lake: Lake object containing: - - - MeteoData (numpy.ndarray): 2D array with columns - for precipitation, evapotranspiration, temperature, - and long-term average temperature. - - Parameters (numpy.ndarray): Lake model parameters. - - CatArea (float): Lake catchment area in km2. - - LakeArea (float): Lake surface area in km2. - - StageDischargeCurve (numpy.ndarray): Stage-discharge - relationship. - - InitialCond (list): Initial condition values. + run: The validated inputs. See :class:`~hapi.runs.DistributedRun`, which + `DistributedRun.from_model(model)` builds and checks. + Lake: The lake record. See :meth:`run_muskingum_with_lake` for the fields it + must carry; this path reads the same ones. - ll_temp (numpy.ndarray, optional): 3D array of long-term - average temperature data. Defaults to None. - q_0 (float, optional): Initial discharge in m3/s. - Defaults to None. + Returns: + SimulationResults: The run's output. Nothing is written to the caller's model; + the entry point in :mod:`hapi.run` is what puts it on `model.results`. """ meteo_data, lake_parameters, outflow_cell = _lake_inputs(Lake) plake = meteo_data[:, 0] @@ -392,39 +360,24 @@ def run_lumped( routes the combined discharge using the provided routing function. - The discharge is converted from mm/timestep to m3/s using - the catchment area and conversion factor. Results are stored - on the Model object as `quz`, `qlz`, `Qsim`, and - `state_variables`. + The discharge is converted from mm/timestep to m3/s using the catchment area and + the period's conversion factor. Args: - Model: Lumped model object containing: - - - data (numpy.ndarray): 2D meteorological data array - with columns for precipitation, - evapotranspiration, temperature, and long-term - average temperature. - - Parameters (numpy.ndarray): Conceptual model - parameters. - - LumpedModel: Conceptual model instance with a - `simulate` method. - - InitialCond (list): Initial state variable values - [sp, sm, uz, lz, wc]. - - q_init (float): Initial discharge value. - - Snow (int): Flag to include snow module (0 or 1). - - CatArea (float): Catchment area in km2. - - conversion_factor (float): Time step conversion - factor (1 for hourly, 0.25 for 15 min, 24 for - daily). - - Maxbas (bool): Whether to use MAXBAS triangular - routing. - - dt (float): Time step duration. - + run: The validated inputs. See :class:`~hapi.runs.LumpedRun`, which + `LumpedRun.from_model(model)` builds and checks. Reads the `(time, 4)` + driver record, the parameter set and the conceptual model setup. Routing (int, optional): Flag to enable routing. Set to 0 to disable, nonzero to enable. Defaults to 0. RoutingFn (callable): Routing function to apply to the discharge hydrograph. Must be callable. + Returns: + SimulationResults: The run's output, with the total discharge in `q_total` and + `routing` set to `RoutingKind.LUMPED`. Nothing is written to the caller's + model; :meth:`~hapi.run.Run.run_lumped` is what indexes `q_total` by the period + and puts the frame on `model.Qsim`. + Raises: TypeError: If `RoutingFn` is not callable when routing is enabled. From 3d01db10c45415692059f3641588d6f2f152046e Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Thu, 10 Sep 2026 22:25:37 +0200 Subject: [PATCH 20/54] fix(wrapper): trim the lumped initial-state slot on every branch, not just the routed ones `Run.run_lumped(model)` -- the entry point called with the `Route=0` it declares as its own default -- raised `ValueError: Length of values (1096) does not match length of index (1095)`. The conceptual model prepends an initial-state slot, so its series is one step longer than the period covers. Both routing branches trimmed it with `q_total[:-1]`; the unrouted path did not, so the total stayed `n + 1` long and could not be indexed by the period. The trim now happens once, above the branches, which is where it belongs: the length is a property of the run, not of which branch it took. The routed paths are unchanged -- they were already trimming the same slot, one line lower. `test_maxbas_routing_convolves_qsim_with_the_last_parameter` had encoded the defect as intended behaviour: it shortened the unrouted reference series itself, and its comment explained that "the unrouted one is a step longer than the index, so only the routed form survives that call". It compares like with like now. Two regression tests cover what nothing covered: the default flag runs and produces a series the period can index, and turning routing on does not change the length. --- src/hapi/wrapper.py | 12 +++- .../catchment/test_maxbas_routing_variants.py | 10 +-- tests/rrm/catchment/test_rrm_catchment.py | 65 +++++++++++++++++++ 3 files changed, 79 insertions(+), 8 deletions(-) diff --git a/src/hapi/wrapper.py b/src/hapi/wrapper.py index ee633f65..728d5ea4 100644 --- a/src/hapi/wrapper.py +++ b/src/hapi/wrapper.py @@ -422,17 +422,23 @@ def run_lumped( # The lumped total discharge is exactly what `q_total` means, so it goes there rather # than onto the catchment as `Qsim`. `Run.run_lumped` is what indexes it by the period # and puts the frame on the model -- so this engine writes nothing outside `results`. - q_total = results.quz + results.qlz + # The conceptual model prepends an initial-state slot, so its series is one step + # longer than the period covers. Trimmed once, here, rather than inside each + # routing branch: both routed branches used to do it and the unrouted one did not, + # so `Run.run_lumped(model)` -- the entry point's own default, `Route=0` -- produced + # an `n + 1` series and then raised `Length of values (1096) does not match length + # of index (1095)` when the period indexed it. + q_total = (results.quz + results.qlz)[:-1] if Routing != 0 and run.parameters.maxbas: route = RoutingFn assert route is not None # noqa: S101 - guarded above - q_total = route(np.array(q_total[:-1]), run.parameters.values[-1]) + q_total = route(np.array(q_total), run.parameters.values[-1]) elif Routing != 0: route = RoutingFn assert route is not None # noqa: S101 - guarded above q_total = route( - np.array(q_total[:-1]), + np.array(q_total), q_total[0], run.parameters.values[-2], run.parameters.values[-1], diff --git a/tests/rrm/catchment/test_maxbas_routing_variants.py b/tests/rrm/catchment/test_maxbas_routing_variants.py index 9c641681..979a4413 100644 --- a/tests/rrm/catchment/test_maxbas_routing_variants.py +++ b/tests/rrm/catchment/test_maxbas_routing_variants.py @@ -233,10 +233,10 @@ def test_maxbas_routing_convolves_qsim_with_the_last_parameter( unrouted series, so compare against the same run left unrouted, convolved independently with the parameter the branch is supposed to use. """ - # Straight to the wrapper: `Run.run_lumped` indexes the series by the period, and the - # unrouted one is a step longer than the index, so only the routed form survives that - # call. The engine leaves its total in `results.q_total`; `Qsim` is what the entry - # point puts on the model. + # Straight to the wrapper for the reference series: the engine leaves its total in + # `results.q_total`, while `Qsim` is what the entry point puts on the model. Both + # forms are now the length of the period -- the initial-state slot is trimmed once, + # for every branch, so the unrouted series no longer has to be shortened here. unrouted = _lumped_model( coello_rrm_date, lumped_meteo_data_path, @@ -257,7 +257,7 @@ def test_maxbas_routing_convolves_qsim_with_the_last_parameter( maxbas = routed.parameters.values[-1] expected = Routing.triangular_routing_1( - np.array(np.asarray(unrouted.results.q_total)[:-1]), maxbas + np.array(np.asarray(unrouted.results.q_total)), maxbas ) # `run_lumped` wraps the routed series in a date-indexed frame; compare the values. actual = np.asarray(routed.Qsim, dtype=float).ravel() diff --git a/tests/rrm/catchment/test_rrm_catchment.py b/tests/rrm/catchment/test_rrm_catchment.py index ff9c8bc7..1312a205 100644 --- a/tests/rrm/catchment/test_rrm_catchment.py +++ b/tests/rrm/catchment/test_rrm_catchment.py @@ -87,6 +87,71 @@ def test_run_lumped( assert len(coello.Qsim) == 10 assert coello.Qsim.columns.to_list() == ["q"] + def test_run_lumped_with_the_default_routing_flag( + self, + coello_rrm_date: list, + lumped_meteo_data_path: str, + coello_AreaCoeff: float, + coello_InitialCond: list, + lumped_parameters_path: str, + coello_Snow: int, + ): + """Test that the entry point runs with the `Route` default it declares. + + Test scenario: + The conceptual model prepends an initial-state slot, so its series is one step + longer than the period. Both routed branches trimmed it and the unrouted one did + not, so `Run.run_lumped(model)` -- the default -- raised `Length of values + (1096) does not match length of index (1095)` when the period indexed the frame. + Nothing exercised the default of a public entry point. + """ + coello = Catchment("rrm", coello_rrm_date[0], coello_rrm_date[1]) + coello.read_lumped_inputs(lumped_meteo_data_path) + coello.read_lumped_model(HBVLumped, coello_AreaCoeff, coello_InitialCond) + coello.read_parameters(lumped_parameters_path, coello_Snow) + + results = Run.run_lumped(coello) + + assert len(results.q_total) == len(coello.period), ( + f"the total must cover the period exactly, got {len(results.q_total)} " + f"against {len(coello.period)}" + ) + assert len(coello.Qsim) == len(coello.period), ( + "the frame the entry point builds is indexed by the period" + ) + + def test_the_routed_and_unrouted_lumped_paths_agree_in_length( + self, + coello_rrm_date: list, + lumped_meteo_data_path: str, + coello_AreaCoeff: float, + coello_InitialCond: list, + lumped_parameters_path: str, + coello_Snow: int, + ): + """Test that turning routing on does not change how long the series is. + + Test scenario: + The trim moved out of the two routing branches and above them, so this pins that + the routed paths still produce exactly what they produced before -- the length is + now a property of the run, not of which branch it took. + """ + + def build() -> Catchment: + model = Catchment("rrm", coello_rrm_date[0], coello_rrm_date[1]) + model.read_lumped_inputs(lumped_meteo_data_path) + model.read_lumped_model(HBVLumped, coello_AreaCoeff, coello_InitialCond) + model.read_parameters(lumped_parameters_path, coello_Snow) + return model + + unrouted = Run.run_lumped(build()) + routed = Run.run_lumped(build(), 1, Routing.muskingum_v) + + assert len(unrouted.q_total) == len(routed.q_total), ( + f"routing must not change the length: {len(unrouted.q_total)} against " + f"{len(routed.q_total)}" + ) + def test_save_lumped_results( self, coello_rrm_date: list, From 24250edf03de537e7cffd3a887300cc0ca2bfd69 Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Thu, 10 Sep 2026 22:31:00 +0200 Subject: [PATCH 21/54] fix(routing): let the routers record the routing they applied `RoutingKind` is presented throughout the package as a property of the arrays -- "the routing is a property of the arrays, so it is read off them instead". It was not. Only `route_muskingum` set it. MAXBAS was labelled by `Wrapper._set_maxbas_output_fields`, one layer *above* the router, and `route_maxbas_by_path_length` -- a public, documented entry point that nothing in the package calls -- set nothing at all. So a caller driving `DistributedRRM` directly, which is the pattern `docs/api/distrrm.md` documents, got MAXBAS-routed arrays still labelled `UNROUTED`, and `outlet_shortcut_valid` answered `True` for them -- offering the outlet-cell shortcut for a scheme that makes a cell a contribution rather than a discharge. Today that fails loudly because `q_total` is still `None`; it is one line of future plumbing away from being silently wrong, which is the failure mode this whole design exists to prevent. The helper moves down into `DistributedRRM._record_maxbas`, beside the routing it describes, and both triangular routers call it. `Wrapper` no longer labels anything it did not route. Two related holes closed with it: * `outlet_shortcut_valid` now excludes `UNROUTED` as well as `MAXBAS`. There is no `q_total` on unrouted results, so there is no cell to read and no shortcut to take. * `extract_discharge` refuses unrouted results by name, rather than picking a branch on that property and failing on `None` several frames in with a message about an array. Six tests cover it: each router records itself, the shortcut is valid for exactly the two schemes where a cell is a discharge, and extracting before routing says which step nobody ran. --- src/hapi/catchment.py | 7 +- src/hapi/results.py | 9 +- src/hapi/rrm/distrrm.py | 44 +++++++- src/hapi/wrapper.py | 50 +-------- tests/rrm/catchment/test_results.py | 163 +++++++++++++++++++++++++++- 5 files changed, 218 insertions(+), 55 deletions(-) diff --git a/src/hapi/catchment.py b/src/hapi/catchment.py index 47692165..965d7593 100644 --- a/src/hapi/catchment.py +++ b/src/hapi/catchment.py @@ -44,7 +44,7 @@ read_rasters, ) from hapi.period import SimulationPeriod -from hapi.results import SimulationResults +from hapi.results import RoutingKind, SimulationResults from hapi.rrm.hbv import HBV from hapi.rrm.hbv_bergestrom92 import HBVBergestrom92 @@ -1062,6 +1062,11 @@ def extract_discharge(self, calculate_metrics=True, factor=None): "there are no results to extract; run the model first, e.g. " "Run.run_distributed(model)" ) + if self.results.routing is RoutingKind.UNROUTED: + raise ValueError( + "these results have not been routed, so there is no hydrograph to extract; " + "call a Run.* entry point rather than DistributedRRM.run_lumped_model alone" + ) if self.results.outlet_shortcut_valid: self.Qsim = pd.DataFrame( diff --git a/src/hapi/results.py b/src/hapi/results.py index fb05c8ac..e5e5c659 100644 --- a/src/hapi/results.py +++ b/src/hapi/results.py @@ -203,11 +203,12 @@ class SimulationResults: def outlet_shortcut_valid(self) -> bool: """bool: Whether a single cell of :attr:`q_total` is the discharge *at* that cell. - True for every scheme except MAXBAS, which routes each cell straight to the outlet - and so makes a cell a contribution rather than a discharge. Reading the outlet cell - of a MAXBAS run under-reports the hydrograph, which is what this guards. + False for MAXBAS, which routes each cell straight to the outlet and so makes a cell + a contribution rather than a discharge -- reading the outlet cell of a MAXBAS run + under-reports the hydrograph, which is what this guards. False for UNROUTED too: + there is no `q_total` yet, so there is no cell to read and no shortcut to take. """ - return self.routing is not RoutingKind.MAXBAS + return self.routing not in (RoutingKind.MAXBAS, RoutingKind.UNROUTED) # ------------------------------------------------------------------ # # narrowing helpers diff --git a/src/hapi/rrm/distrrm.py b/src/hapi/rrm/distrrm.py index 5c695cef..55498390 100644 --- a/src/hapi/rrm/distrrm.py +++ b/src/hapi/rrm/distrrm.py @@ -187,13 +187,50 @@ def route_muskingum(run: DistributedRun, results: SimulationResults) -> None: # cell and the outlet-cell shortcut in `extract_discharge` is valid. results.routing = RoutingKind.MUSKINGUM + @staticmethod + def _record_maxbas(results: SimulationResults) -> None: + """Fill the per-cell output fields after a triangular (MAXBAS) routing pass. + + `results.save` and `results.animate` read `q_total`, `quz_routed` and + `qlz_translated` for their discharge options. This used to live on `Wrapper`, which + meant the routing kind was recorded by the layer *above* the router: a caller + driving `DistributedRRM` directly -- the pattern `docs/api/distrrm.md` documents -- + got MAXBAS-routed arrays still labelled `RoutingKind.UNROUTED`, and + `outlet_shortcut_valid` then answered for a scheme that had not run. The routing is + meant to be a property of the arrays, so the method that applies it is what records + it. + + MAXBAS routes each cell's upper zone straight to the outlet with that cell's own + `maxbas`, in place, and applies no cell-to-cell translation to the lower zone. So the + routed/translated fields *are* the per-cell arrays, and their sum is the per-cell + contribution to the outlet hydrograph -- `np.nansum(q_total[:, :, i])` reproduces + `qout[i]`. That differs from the Muskingum path, where the fields accumulate + downstream and `q_total` at the outlet cell *is* the outlet discharge. + + `quz_routed` / `qlz_translated` alias `quz` / `qlz` rather than copying them: they + hold the same data, and a copy would double the memory of a + `(rows, cols, time_steps)` array for no gain. They are outputs, so nothing downstream + writes through the alias -- but the alias is visible (`results.quz_routed is + results.quz`), so an in-place edit of one changes the other. + + Args: + results: The results whose `quz` / `qlz` have just been routed. Mutated in place. + """ + results.quz_routed = results.quz + results.qlz_translated = results.qlz + results.q_total = results.qlz + results.quz + # Marks the outlet-cell shortcut in `extract_discharge` as invalid for these + # results, via `SimulationResults.outlet_shortcut_valid`. + results.routing = RoutingKind.MAXBAS + @staticmethod def route_maxbas(run: DistributedRun, results: SimulationResults) -> None: """Route discharge to the outlet using a triangular function. Applies triangular (MAXBAS) routing to each cell's upper-zone discharge independently, reading the MAXBAS parameter from the last column of the parameter array. `results.quz` - is modified in place. + is modified in place, then the per-cell output fields are filled and + `RoutingKind.MAXBAS` recorded by :meth:`_record_maxbas`. Args: run: The validated inputs. @@ -208,6 +245,7 @@ def route_maxbas(run: DistributedRun, results: SimulationResults) -> None: quz[x, y, :] = routing.triangular_routing_1( quz[x, y, :], Maxbas[x, y] ) + DistributedRRM._record_maxbas(results) @staticmethod def route_maxbas_by_path_length( @@ -217,7 +255,8 @@ def route_maxbas_by_path_length( Like :meth:`route_maxbas`, but each cell's MAXBAS is rescaled by its flow path length, so cells farther from the outlet are attenuated more. `results.quz` is modified in - place. + place, then the per-cell output fields are filled and `RoutingKind.MAXBAS` recorded + by :meth:`_record_maxbas`. Args: run: The validated inputs, whose `flow_path_length` supplies the raster. @@ -253,3 +292,4 @@ def route_maxbas_by_path_length( quz[x, y, :] = routing.triangular_routing_2( quz[x, y, :], NormalizedFPL[x, y] ) + DistributedRRM._record_maxbas(results) diff --git a/src/hapi/wrapper.py b/src/hapi/wrapper.py index 728d5ea4..4a4280ae 100644 --- a/src/hapi/wrapper.py +++ b/src/hapi/wrapper.py @@ -191,40 +191,6 @@ def run_muskingum_with_lake(run: DistributedRun, Lake: Lake) -> SimulationResult distrrm.route_muskingum(run, results) return results - @staticmethod - def _set_maxbas_output_fields(results: SimulationResults) -> None: - """Fill the distributed output fields after a triangular (MAXBAS) run. - - `results.save` and `results.animate` read `q_total`, - `quz_routed` and `qlz_translated` for their discharge options. Only - :meth:`DistRRM.route_muskingum` (the Muskingum path) used to set them, so - after a MAXBAS run they stayed `None` and every discharge option raised - `TypeError: 'NoneType' object is not subscriptable`. - - MAXBAS routes each cell's upper zone straight to the outlet with that - cell's own `maxbas`, in place, and applies no cell-to-cell translation - to the lower zone. So the routed/translated fields *are* the per-cell - arrays, and their sum is the per-cell contribution to the outlet - hydrograph — `np.nansum(q_total[:, :, i])` reproduces `qout[i]`. That - differs from the Muskingum path, where the fields accumulate downstream - and `q_total` at the outlet cell *is* the outlet discharge. - - `quz_routed` / `qlz_translated` alias `quz` / `qlz` rather than - copying them: they hold the same data, and a copy would double the memory - of a `(rows, cols, time_steps)` array for no gain. They are outputs, so - nothing downstream writes through the alias. - - Args: - results: The results whose `quz` / `qlz` have been routed by - :meth:`~hapi.rrm.distrrm.DistributedRRM.route_maxbas`. Mutated in place. - """ - results.quz_routed = results.quz - results.qlz_translated = results.qlz - results.q_total = results.qlz + results.quz - # Marks the outlet-cell shortcut in `extract_discharge` as invalid for these - # results, via `SimulationResults.outlet_shortcut_valid`. - results.routing = RoutingKind.MAXBAS - @staticmethod def run_maxbas(run: DistributedRun) -> SimulationResults: """Run the distributed RRM with triangular function-1 routing. @@ -237,10 +203,10 @@ def run_maxbas(run: DistributedRun) -> SimulationResults: The output discharge is computed as the sum of routed upper zone and unrouted lower zone discharge across all cells. - Also fills the per-cell output fields (`q_total`, `quz_routed`, - `qlz_translated`) via :meth:`_set_maxbas_output_fields`, so the - discharge options of `results.save` / `results.animate` - work on this path; see that method for the MAXBAS semantics. + :meth:`~hapi.rrm.distrrm.DistributedRRM.route_maxbas` fills the per-cell output + fields (`q_total`, `quz_routed`, `qlz_translated`) and records + `RoutingKind.MAXBAS`, so the discharge options of `results.save` / + `results.animate` work on this path; see that method for the MAXBAS semantics. Args: run: The validated inputs. See :class:`~hapi.runs.DistributedRun`, which @@ -255,8 +221,6 @@ def run_maxbas(run: DistributedRun) -> SimulationResults: distrrm.route_maxbas(run, results) - Wrapper._set_maxbas_output_fields(results) - steps = run.meteo.simulation_steps qlz1 = np.array( [np.nansum(results.qlz[:, :, i]) for i in range(steps)] @@ -325,12 +289,10 @@ def run_maxbas_with_lake(run: DistributedRun, Lake: Lake) -> SimulationResults: # subcatchment results = distrrm.run_lumped_model(run) + # `route_maxbas` fills the subcatchment fields only: the lake is a lumped inflow + # with no spatial extent, so it enters `qout` below but never `q_total`. distrrm.route_maxbas(run, results) - # Subcatchment fields only: the lake is a lumped inflow with no spatial - # extent, so it enters `qout` below but never `q_total`. - Wrapper._set_maxbas_output_fields(results) - steps = run.meteo.simulation_steps qlz1 = np.array( [np.nansum(results.qlz[:, :, i]) for i in range(steps)] diff --git a/tests/rrm/catchment/test_results.py b/tests/rrm/catchment/test_results.py index b61ffaf1..b2d0774e 100644 --- a/tests/rrm/catchment/test_results.py +++ b/tests/rrm/catchment/test_results.py @@ -12,6 +12,7 @@ from __future__ import annotations import datetime as dt +from copy import copy import numpy as np import pandas as pd @@ -30,7 +31,7 @@ @pytest.fixture(scope="module") -def distributed_run( +def built_catchment( coello_start_date: str, coello_end_date: str, coello_prec_path: str, @@ -41,11 +42,12 @@ def distributed_run( coello_dist_parameters_muskingum: str, coello_cat_area: int, coello_initial_cond: list, -) -> DistributedRun: - """A validated distributed Coello run, with no engine having touched it yet. + coello_gauges_table: str, +) -> Catchment: + """A fully built distributed Coello catchment, gauge table included. Returns: - DistributedRun: The narrowed run. + Catchment: The builder, with every reader run and no run behind it. """ model = Catchment( "coello", @@ -67,6 +69,39 @@ def distributed_run( model.flow_network = FlowNetwork.from_rasters(coello_acc_path, coello_fd_path) model.read_parameters(coello_dist_parameters_muskingum, False) model.read_lumped_model(HBVLumped, coello_cat_area, coello_initial_cond) + model.read_gauge_table(coello_gauges_table, coello_acc_path) + return model + + +@pytest.fixture(scope="module") +def distributed_run(built_catchment: Catchment) -> DistributedRun: + """A validated distributed Coello run, with no engine having touched it yet. + + Args: + built_catchment: The builder it narrows. + + Returns: + DistributedRun: The narrowed run. + """ + return DistributedRun.from_model(built_catchment) + + +@pytest.fixture(scope="module") +def maxbas_run( + built_catchment: Catchment, + coello_dist_parameters_maxbas: str, +) -> DistributedRun: + """The same catchment narrowed with a MAXBAS parameter set. + + Args: + built_catchment: The builder, whose Muskingum parameters are replaced here. + coello_dist_parameters_maxbas: Parameter cube carrying a MAXBAS trailing column. + + Returns: + DistributedRun: A run the triangular routers can actually route. + """ + model = copy(built_catchment) + model.read_parameters(coello_dist_parameters_maxbas, False, maxbas=True) return DistributedRun.from_model(model) @@ -155,6 +190,126 @@ def test_a_lumped_run_cannot_be_animated(self, lumped_results: SimulationResults lumped_results.animate("2012-06-14", "2012-06-20", option=1) +class TestTheRouterRecordsTheRoutingItApplied: + """`RoutingKind` is meant to be a property of the arrays, not of who called whom.""" + + def test_route_maxbas_records_itself(self, maxbas_run: DistributedRun): + """Test that the public MAXBAS router labels the arrays it just routed. + + Args: + maxbas_run: A validated run carrying a MAXBAS parameter set. + + Test scenario: + The labelling used to sit in `Wrapper`, one layer above the router. Driving + `DistributedRRM` directly -- the pattern `docs/api/distrrm.md` documents -- + then produced MAXBAS-routed arrays still labelled `UNROUTED`, so every consumer + that asks the arrays what happened to them got the wrong answer. + """ + results = DistributedRRM.run_lumped_model(maxbas_run) + assert results.routing is RoutingKind.UNROUTED, "nothing has routed them yet" + + DistributedRRM.route_maxbas(maxbas_run, results) + + assert results.routing is RoutingKind.MAXBAS, ( + f"the router must record what it applied, got {results.routing}" + ) + for field in ("q_total", "quz_routed", "qlz_translated"): + assert getattr(results, field) is not None, ( + f"{field} backs the discharge options and must be filled by the router" + ) + + def test_route_maxbas_by_path_length_records_itself( + self, maxbas_run: DistributedRun + ): + """Test that the path-length variant records the routing too. + + Args: + maxbas_run: A validated run, rebuilt here with a path-length raster. + + Test scenario: + This entry point is public and documented, and nothing inside the package calls + it -- so it was the one router where nothing set the routing at all. + """ + model = maxbas_run + rows, cols = model.flow_network.shape + # A gradient, masked to the domain: the routing normalises by (max - min), so a + # constant raster would divide by zero, and NaN marks the cells outside the basin. + gradient = np.arange(rows * cols, dtype=float).reshape(rows, cols) + gradient[np.isnan(model.flow_network.flow_acc_arr)] = np.nan + with_fpl = DistributedRun( + period=model.period, + meteo=model.meteo, + flow_network=model.flow_network, + parameters=model.parameters, + model_setup=model.model_setup, + flow_path_length=gradient, + ) + results = DistributedRRM.run_lumped_model(with_fpl) + + DistributedRRM.route_maxbas_by_path_length(with_fpl, results) + + assert results.routing is RoutingKind.MAXBAS, ( + f"the path-length router must record what it applied, got {results.routing}" + ) + assert results.q_total is not None, "the per-cell fields must be filled" + + +class TestTheOutletShortcut: + """Reading the outlet cell only means something for a scheme that accumulates.""" + + @pytest.mark.parametrize( + "routing, valid", + [ + (RoutingKind.MUSKINGUM, True), + (RoutingKind.LUMPED, True), + (RoutingKind.MAXBAS, False), + (RoutingKind.UNROUTED, False), + ], + ) + def test_the_shortcut_is_valid_only_where_a_cell_is_a_discharge( + self, routing: RoutingKind, valid: bool + ): + """Test which routing kinds allow the outlet cell to be read as the hydrograph. + + Args: + routing: The routing the arrays carry. + valid: Whether the outlet-cell shortcut holds for it. + + Test scenario: + MAXBAS makes a cell a *contribution* rather than a discharge, so reading the + outlet cell under-reports. UNROUTED has no `q_total` at all -- the property used + to answer `True` for it, offering the shortcut for a scheme that never ran. + """ + cube = np.zeros((2, 3, 4), dtype="float32") + results = SimulationResults(routing, cube, cube, None) + + assert results.outlet_shortcut_valid is valid, ( + f"{routing} should give outlet_shortcut_valid={valid}" + ) + + def test_extracting_from_unrouted_results_names_the_missing_step( + self, built_catchment: Catchment + ): + """Test that extracting a hydrograph before routing names the step nobody ran. + + Args: + built_catchment: A distributed catchment with its gauge table read. + + Test scenario: + `run_lumped_model` alone leaves `q_total` as `None`. Reaching + `extract_discharge` with it used to depend on `outlet_shortcut_valid` answering + for a scheme that never ran, and either branch then failed on `None` with a + message about an array rather than about the routing nobody applied. + """ + catchment = copy(built_catchment) + catchment.results = DistributedRRM.run_lumped_model( + DistributedRun.from_model(built_catchment) + ) + + with pytest.raises(ValueError, match="have not been routed"): + catchment.extract_discharge() + + class TestUnfilledFieldsAreNamed: """A routed field is `None` until its routing step runs, and that is a real state.""" From b214d9af82a3361e0ac08521e43341c9838323d5 Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Thu, 10 Sep 2026 22:33:37 +0200 Subject: [PATCH 22/54] fix(calibration): let a wrongly-wired objective function reach the caller `opt_fun` catches `TypeError` from the objective call and re-raises it as the "objective function you have entered needs more inputs" error. That `raise` sat *inside* the `try` that classifies a trial as numerically infeasible, so the `ValueError` it raised was caught by `except Exception` one line later and scored `np.nan` with `fail=1`. The result: a caller who wired up an objective with the wrong signature got a full Harmony Search over an all-`nan` landscape and one warning per trial, never the message the constant was written for. Narrowing the bare `except:` to `except Exception:` earlier in this branch fixed the `KeyboardInterrupt` problem but not this one -- the error is a `ValueError`, and `Exception` catches it either way. `ObjectiveFunctionArityError` gives it a type of its own so it can travel through that handler, and each of the three entry points re-raises it above the numerical case. It stays a `ValueError` subclass, so anything already catching `ValueError` around a calibration is unaffected. Two tests, one for each side of the handler: an objective of the wrong arity now reaches the caller, and an objective that fails on the *values* is still scored infeasible and still lets the calibration continue -- which is what the handler is for, and what a careless re-raise would have broken. --- src/hapi/calibration.py | 27 ++++++++- .../test_calibration_distributed.py | 55 ++++++++++++++++++- 2 files changed, 78 insertions(+), 4 deletions(-) diff --git a/src/hapi/calibration.py b/src/hapi/calibration.py index 7c5f5fe5..ec2026f4 100644 --- a/src/hapi/calibration.py +++ b/src/hapi/calibration.py @@ -32,6 +32,17 @@ ) +class ObjectiveFunctionArityError(ValueError): + """The objective function was called with fewer arguments than it declares. + + Its own type, rather than a bare `ValueError`, because it has to travel *through* the + handler that classifies a trial as numerically infeasible. Raised as a `ValueError` it + was caught by that handler one line later and scored `np.nan`, so a caller who wired up + an objective of the wrong arity got a full Harmony Search over an all-`nan` landscape + and a warning per trial instead of the error this message was written for. + """ + + def _check_optimization_args(api_obj_args: Any, api_solve_args: Any) -> None: """Check the two argument bundles the optimizer is handed are mappings. @@ -489,7 +500,7 @@ def opt_fun(par): except TypeError as e: # the objective function received fewer inputs than it needs - raise ValueError(OBJECTIVE_FN_ARGS_ERROR) from e + raise ObjectiveFunctionArityError(OBJECTIVE_FN_ARGS_ERROR) from e # print error if print_error != 0: @@ -497,6 +508,10 @@ def opt_fun(par): print(par) fail = 0 + except ObjectiveFunctionArityError: + # Not a bad parameter set: the objective function itself is wired up wrong, + # and every trial would fail the same way. Let it out. + raise except Exception as exc: # A genuine numerical failure for this candidate. Narrowed from a bare # `except`, which also caught KeyboardInterrupt -- so a long calibration @@ -633,7 +648,7 @@ def opt_fun(par): ) except TypeError as e: # the objective function received fewer inputs than it needs - raise ValueError(OBJECTIVE_FN_ARGS_ERROR) from e + raise ObjectiveFunctionArityError(OBJECTIVE_FN_ARGS_ERROR) from e # print error if print_error != 0: @@ -641,6 +656,9 @@ def opt_fun(par): print(par) fail = 0 + except ObjectiveFunctionArityError: + # See run_calibration: a wrongly-wired objective is not a bad candidate. + raise except Exception as exc: # See run_calibration: narrowed from a bare `except`. logger.warning(f"trial failed, scoring it infeasible: {exc!r}") @@ -787,7 +805,7 @@ def opt_fun(par): ] except TypeError as e: # the objective function received fewer inputs than it needs - raise ValueError(OBJECTIVE_FN_ARGS_ERROR) from e + raise ObjectiveFunctionArityError(OBJECTIVE_FN_ARGS_ERROR) from e if print_error != 0: print( @@ -795,6 +813,9 @@ def opt_fun(par): ) # print(par) fail = 0 + except ObjectiveFunctionArityError: + # See run_calibration: a wrongly-wired objective is not a bad candidate. + raise except Exception as exc: # A genuine numerical failure for this candidate. Narrowed from a bare # `except`, which also caught KeyboardInterrupt -- so a long calibration diff --git a/tests/rrm/calibration/test_calibration_distributed.py b/tests/rrm/calibration/test_calibration_distributed.py index 6b64f101..818cb473 100644 --- a/tests/rrm/calibration/test_calibration_distributed.py +++ b/tests/rrm/calibration/test_calibration_distributed.py @@ -13,7 +13,7 @@ from pandas import DataFrame from hapi import calibration as calibration_module -from hapi.calibration import Calibration +from hapi.calibration import Calibration, ObjectiveFunctionArityError from hapi.catchment import Catchment from hapi.conceptual import ParameterBounds from hapi.inputs import FlowNetwork, MeteoInputs @@ -229,6 +229,59 @@ def test_stores_the_optimizer_result_on_the_instance( ), ) + def test_a_wrongly_wired_objective_reaches_the_caller( + self, gauged_calibration: Calibration, stub_optimizer: dict, spatial_var_stub + ): + """Test that an objective of the wrong arity raises instead of scoring `nan`. + + Test scenario: + `opt_fun` catches `TypeError` from the objective call and re-raises it as the + "needs more inputs" error -- but that `raise` sat inside the `try` that + classifies a trial as numerically infeasible, so the error it raised was caught + one line later and turned into `(nan, [], 1)`. A caller who wired up an objective + with the wrong signature got a full Harmony Search over an all-`nan` landscape + and a warning per trial, never the message the constant was written for. It + carries its own type now so it can travel through that handler. + """ + coello = gauged_calibration + coello.bounds = ParameterBounds(np.zeros(12), np.ones(12)) + + def needs_four_arguments(observed, simulated, gauges, extra): + """Take more arguments than `run_calibration` passes.""" + return 0.0 + + coello.read_objective_function(needs_four_arguments, []) + + with pytest.raises(ObjectiveFunctionArityError, match="needs more inputs"): + coello.run_calibration(spatial_var_stub, _optimization_args()) + + def test_a_numerically_failing_trial_is_still_scored_infeasible( + self, gauged_calibration: Calibration, stub_optimizer: dict, spatial_var_stub + ): + """Test that letting the arity error out did not stop real failures being caught. + + Test scenario: + The point of the handler is that one bad candidate does not end a calibration. + Adding a re-raise above it risks turning every numerical failure into a crash, + so this pins the other side: an objective that blows up on the *values* is still + scored infeasible and the optimiser still runs. + """ + coello = gauged_calibration + coello.bounds = ParameterBounds(np.zeros(12), np.ones(12)) + + def divides_by_zero(observed, gauges): + """Fail on the numbers, not on the signature.""" + raise ZeroDivisionError("no discharge at this gauge") + + coello.read_objective_function(divides_by_zero, []) + + coello.run_calibration(spatial_var_stub, _optimization_args()) + + assert "n_vars" in stub_optimizer, ( + "a failing candidate must not stop the calibration; the optimiser should still " + "have been driven" + ) + def test_rejects_meteo_that_does_not_cover_the_grid( self, gauged_calibration: Calibration, stub_optimizer: dict, spatial_var_stub ): From 9d96d151e2db100a15198f7c1c5cfb0cca440f07 Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Thu, 10 Sep 2026 22:37:44 +0200 Subject: [PATCH 23/54] fix(runs): hold the flow-path-length raster to the catchment grid `DistributedRun.from_model` is documented as the single seam where everything checkable is checked, and it checks the drivers, the parameter cube and the river geometry against the grid. The flow-path-length raster was carried in with a bare `getattr` and compared to nothing -- `read_flow_path_length` does not compare it either, having deliberately stopped deriving `rows`/`cols` from it. `route_maxbas_by_path_length` then indexes it by `flow_network.rows`/`cols`, so a raster on a different grid either raised `IndexError` several frames inside that loop or, if it was larger, quietly read the wrong cells for every cell of the catchment. Three shapes are refused and the matching one is accepted, so the guard is shown doing both halves of its job. --- src/hapi/runs.py | 19 +++++++++-- tests/rrm/catchment/test_run_narrowing.py | 39 +++++++++++++++++++++++ 2 files changed, 55 insertions(+), 3 deletions(-) diff --git a/src/hapi/runs.py b/src/hapi/runs.py index 313f27ad..3020a8b1 100644 --- a/src/hapi/runs.py +++ b/src/hapi/runs.py @@ -101,9 +101,9 @@ def __post_init__(self): """Check the inputs agree with each other and with the grid. Raises: - ValueError: The drivers or the parameters do not cover the grid, the river geometry - does not, or a cell skip was asked for with no geometry to identify the river - cells. + ValueError: The drivers, the parameters, the river geometry or the flow-path-length + raster do not cover the grid, or a cell skip was asked for with no geometry to + identify the river cells. """ rows, cols = self.flow_network.rows, self.flow_network.cols @@ -122,6 +122,19 @@ def __post_init__(self): ): raise ValueError(GRID_MISMATCH_ERROR) + # `route_maxbas_by_path_length` indexes this by the flow network's rows and cols, + # so a raster on a different grid either raises deep inside that loop or -- if it is + # larger -- quietly reads the wrong cells. It was carried in with a bare `getattr` + # and was the only input this seam did not check. + if self.flow_path_length is not None: + shape = np.shape(self.flow_path_length) + if shape != (rows, cols): + raise ValueError( + f"the flow-path-length raster is {shape} but the catchment grid is " + f"({rows}, {cols}); read it from a raster aligned to the " + f"flow-accumulation grid" + ) + if self.skip_hydraulic_cells and self.river_geometry is None: raise ValueError( "skipping the hydraulic cells needs the river geometry to identify them, " diff --git a/tests/rrm/catchment/test_run_narrowing.py b/tests/rrm/catchment/test_run_narrowing.py index f2e83091..6c686dc2 100644 --- a/tests/rrm/catchment/test_run_narrowing.py +++ b/tests/rrm/catchment/test_run_narrowing.py @@ -156,6 +156,45 @@ def test_a_skip_without_geometry_is_refused(self, built: Catchment): with pytest.raises(ValueError, match="read_river_geometry"): DistributedRun.from_model(built, skip_hydraulic_cells=True) + @pytest.mark.parametrize("delta", [(1, 0), (0, 1), (-1, -1)]) + def test_a_flow_path_length_raster_off_the_grid_is_refused( + self, built: Catchment, delta: tuple[int, int] + ): + """Test that the path-length raster is held to the grid like every other input. + + Args: + built: A fully built distributed catchment. + delta: Row and column offsets applied to the raster's shape. + + Test scenario: + `route_maxbas_by_path_length` indexes this raster by the flow network's rows and + cols. It was the one input carried into the run with a bare `getattr` and never + compared to the grid, so a mismatched raster either raised `IndexError` deep in + that loop or -- when larger -- quietly read the wrong cells. + """ + rows, cols = built.flow_network.shape + built.flow_path_length_arr = np.ones((rows + delta[0], cols + delta[1])) + + with pytest.raises(ValueError, match="flow-path-length raster"): + DistributedRun.from_model(built) + + def test_a_flow_path_length_raster_on_the_grid_is_accepted(self, built: Catchment): + """Test that the check admits a raster that does match the grid. + + Args: + built: A fully built distributed catchment. + + Test scenario: + The other half of the guard: a shape check that refused everything would pass the + test above and break the only entry point that reads this raster. + """ + rows, cols = built.flow_network.shape + built.flow_path_length_arr = np.ones((rows, cols)) + + run = DistributedRun.from_model(built) + + assert run.flow_path_length is not None, "the raster must reach the run" + def test_geometry_off_the_catchment_grid_is_refused(self, built: Catchment): """Test that geometry on a different grid than the catchment raises. From 445c76b255acdb760f3f34c1c3956e54c42d4152 Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Thu, 10 Sep 2026 22:37:44 +0200 Subject: [PATCH 24/54] fix(run): require the fourth lake driver the wrapper actually reads The guard admitted any lake record with at least three columns and its message named "rain, ET, and Temp". Both lake wrappers then read `meteo_data[:, 3]` for the long-term average temperature, so a three-column record passed validation and raised `IndexError` inside the run -- naming a column index rather than the driver nobody supplied. The check and the message now ask for the four columns the code reads. The test is parametrised over two and three columns: three is the width that regressed, two is kept so the guard is still shown refusing what it always refused, and both now assert the engine is never reached. --- src/hapi/run.py | 8 +++++-- tests/rrm/catchment/test_run_validation.py | 27 ++++++++++++++++------ 2 files changed, 26 insertions(+), 9 deletions(-) diff --git a/src/hapi/run.py b/src/hapi/run.py index b1b745a1..be10f8a4 100644 --- a/src/hapi/run.py +++ b/src/hapi/run.py @@ -57,9 +57,13 @@ def _check_lake_meteo(run: DistributedRun, lake: LakeType) -> None: "Lake meteorological data has to have the same length as the distributed " "raster data" ) - if np.shape(meteo_data)[1] < 3: + # Four, not three: both lake wrappers read `meteo_data[:, 3]` for the long-term + # average, so a three-column record passed this check and then raised `IndexError` + # inside the run, naming a column index rather than the missing driver. + if np.shape(meteo_data)[1] < 4: raise ValueError( - "Lake Meteo data has to have at least three columns of rain, ET, and Temp" + "Lake meteorological data has to have at least four columns: rain, ET, " + "temperature, and the long-term average temperature" ) diff --git a/tests/rrm/catchment/test_run_validation.py b/tests/rrm/catchment/test_run_validation.py index 848b3926..dff174ac 100644 --- a/tests/rrm/catchment/test_run_validation.py +++ b/tests/rrm/catchment/test_run_validation.py @@ -24,7 +24,7 @@ class _LakeStub: """Stand-in for `Lake` carrying only the `MeteoData` the validation reads.""" - def __init__(self, time_steps: int, columns: int = 3): + def __init__(self, time_steps: int, columns: int = 4): self.MeteoData = np.ones((time_steps, columns)) @@ -229,20 +229,33 @@ def test_rejects_a_lake_record_of_the_wrong_length( "the wrapper must not run against a mismatched lake record" ) + @pytest.mark.parametrize("columns", [2, 3]) def test_rejects_a_lake_record_missing_a_column( - self, coello_loaded: Catchment, spied_wrapper: dict + self, coello_loaded: Catchment, spied_wrapper: dict, columns: int ): - """Test that a lake record without all three drivers is refused. + """Test that a lake record without all four drivers is refused. + + Args: + coello_loaded: A distributed catchment with every input read. + spied_wrapper: Records whether an engine was reached. + columns: Width of the lake record under test. Test scenario: - The lake model reads rain, ET and temperature by position, so fewer than three - columns cannot be interpreted. + The guard asked for three columns while both lake wrappers read + `meteo_data[:, 3]` for the long-term average, so a three-column record passed + validation and then raised `IndexError` inside the run -- naming a column index + rather than the driver nobody supplied. Three is the case that regressed; two is + kept so the guard is still shown refusing what it always refused. """ - lake = _LakeStub(coello_loaded.meteo.time_steps, columns=2) + lake = _LakeStub(coello_loaded.meteo.time_steps, columns=columns) - with pytest.raises(ValueError, match="three columns"): + with pytest.raises(ValueError, match="four columns"): Run.run_distributed_with_lake(coello_loaded, lake) + assert not spied_wrapper, ( + "the engine must not be reached with a record it cannot read" + ) + def test_a_flow_direction_grid_of_the_wrong_shape_cannot_be_installed( self, coello_loaded: Catchment ): From 34f2f4d92cb4e129346974ecfa4bde6a872a5348 Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Thu, 10 Sep 2026 22:37:58 +0200 Subject: [PATCH 25/54] fix(catchment): give the outlet hydrograph the same length on every routing path `SimulationResults.qout` is documented as one thing -- "the outlet hydrograph" -- but its length depended on how the run had been routed. The MAXBAS and lake paths trim the conceptual model's leading initial-state slot and return `len(period)` values; the Muskingum path read the outlet cell of `q_total` whole and returned one more. Anything indexing `qout` by the period therefore worked on one routing path and raised on the other. The Muskingum branch trims like the rest, and the field's docstring now states the length rather than leaving it to be discovered. A test pins that it covers the period exactly. --- src/hapi/catchment.py | 5 +++- src/hapi/results.py | 6 +++-- .../test_extract_discharge_distributed.py | 23 +++++++++++++++++++ 3 files changed, 31 insertions(+), 3 deletions(-) diff --git a/src/hapi/catchment.py b/src/hapi/catchment.py index 965d7593..4a511213 100644 --- a/src/hapi/catchment.py +++ b/src/hapi/catchment.py @@ -1082,7 +1082,10 @@ def extract_discharge(self, calculate_metrics=True, factor=None): # Muskingum accumulates downstream, so the outlet cell of `q_total` is the # outlet hydrograph. The engine cannot set this itself: finding the outlet # needs the gauge table, which is an analysis input, not a run input. - self.results.qout = self.results.q_total[outlet_x, outlet_y, :] + # Trimmed like every other path: `q_total` carries the conceptual model's + # initial-state slot, so the untrimmed form made `qout` a step longer here than + # on the MAXBAS and lake paths, for a field documented as one hydrograph. + self.results.qout = self.results.q_total[outlet_x, outlet_y, :-1] for i in range(len(self.GaugesTable)): x_ind = int(self.GaugesTable.loc[self.GaugesTable.index[i], "cell_row"]) diff --git a/src/hapi/results.py b/src/hapi/results.py index e5e5c659..59aabbaf 100644 --- a/src/hapi/results.py +++ b/src/hapi/results.py @@ -127,8 +127,10 @@ class SimulationResults: qlz_translated: Lower-zone discharge after translation. `None` until then. q_total: `quz_routed + qlz_translated`. Read it through :attr:`outlet_shortcut_valid` rather than assuming what a cell means. - qout: The outlet hydrograph, when the run computed one. The MAXBAS paths sum over the - domain and set it directly; the Muskingum paths leave it `None` for + qout: The outlet hydrograph, when the run computed one, and always `len(period)` + long -- the conceptual model's leading initial-state slot is dropped on every + path that fills it. The MAXBAS paths sum over the domain and set it directly; + the Muskingum paths leave it `None` for :meth:`~hapi.catchment.Catchment.extract_discharge` to read off the outlet cell, which needs the gauge table the engine does not have. run: The validated inputs these arrays came from, carried as provenance. It is what diff --git a/tests/rrm/catchment/test_extract_discharge_distributed.py b/tests/rrm/catchment/test_extract_discharge_distributed.py index ffb50c1e..7c407e93 100644 --- a/tests/rrm/catchment/test_extract_discharge_distributed.py +++ b/tests/rrm/catchment/test_extract_discharge_distributed.py @@ -56,6 +56,29 @@ def coello_muskingum_run( return coello +def test_qout_covers_the_period_exactly(coello_muskingum_run: Catchment): + """Test that the outlet hydrograph is as long as the period, not a step longer. + + Args: + coello_muskingum_run: Distributed Coello catchment with a completed Muskingum run. + + Test scenario: + `q_total` carries the conceptual model's leading initial-state slot, so reading the + outlet cell whole gave a `qout` one step longer than the MAXBAS and lake paths + produce -- for a field documented as one thing, "the outlet hydrograph". Anything + indexing it by the period worked on one routing path and not the other. + """ + # The Muskingum path leaves `qout` for this call to fill: finding the outlet needs the + # gauge table, which the engine does not have. + coello_muskingum_run.extract_discharge() + qout = coello_muskingum_run.results.qout + + assert len(qout) == len(coello_muskingum_run.period), ( + f"qout must cover the period: {len(qout)} against " + f"{len(coello_muskingum_run.period)}" + ) + + def test_extract_discharge_distributed_metrics(coello_muskingum_run: Catchment): """The distributed branch computes all seven metrics for every gauge. From a7f9729bcfbef5b515a716bdc5ceed28830d81d3 Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Thu, 10 Sep 2026 22:37:59 +0200 Subject: [PATCH 26/54] fix(conceptual): check the parameter bounds against the width they declare `ParameterBounds.__post_init__` compared `len(lower)` with `len(upper)` and stopped there, while its own docstring says "the bounds are where the configuration enters, and every trial vector the optimiser produces is checked against it". The bounds themselves were not checked against `PARAMETER_COUNTS[(snow, maxbas)]`. So a ten-value bound list with `snow=False, maxbas=False` built fine, and the mismatch surfaced once per trial from `ParameterSet.__post_init__` -- after the whole optimisation problem had been declared and the optimiser started, from inside the objective, rather than at the call that got it wrong. It calls the same `validate_parameter_count` every trial vector goes through. --- src/hapi/conceptual.py | 8 ++++- tests/rrm/calibration/test_rrm_calibration.py | 30 +++++++++++++++++++ 2 files changed, 37 insertions(+), 1 deletion(-) diff --git a/src/hapi/conceptual.py b/src/hapi/conceptual.py index a9c78e11..2488a122 100644 --- a/src/hapi/conceptual.py +++ b/src/hapi/conceptual.py @@ -293,13 +293,19 @@ def __post_init__(self): """Check the two bounds describe the same parameters. Raises: - ValueError: The bounds are different lengths. + ValueError: The bounds are different lengths, or do not hold the number of + parameters `(snow, maxbas)` calls for. """ if len(self.lower) != len(self.upper): raise ValueError( f"the length of UB should be the same as LB, got {len(self.upper)} and " f"{len(self.lower)}" ) + # The same rule every trial vector is held to. Checked here as well because this is + # where the configuration enters: a mismatch used to surface once per trial from + # `ParameterSet`, after the whole optimisation problem had been declared and the + # optimiser started, rather than at the call that got it wrong. + validate_parameter_count(self.lower, self.snow, self.maxbas) object.__setattr__(self, "lower", np.array(self.lower)) object.__setattr__(self, "upper", np.array(self.upper)) diff --git a/tests/rrm/calibration/test_rrm_calibration.py b/tests/rrm/calibration/test_rrm_calibration.py index fc5423ac..cdfb1743 100644 --- a/tests/rrm/calibration/test_rrm_calibration.py +++ b/tests/rrm/calibration/test_rrm_calibration.py @@ -25,6 +25,36 @@ def test_read_parameters_bounds( assert isinstance(Coello.bounds.maxbas, bool) +@pytest.mark.parametrize( + "width, snow, maxbas", + [(10, False, False), (12, True, True), (16, False, True)], +) +def test_read_parameters_bounds_refuses_a_width_the_configuration_does_not_call_for( + coello_rrm_date: list, width: int, snow: bool, maxbas: bool +): + """Test that bounds of the wrong width are refused where they enter. + + Args: + coello_rrm_date: Start and end dates for the model. + width: Number of bound values supplied. + snow: Whether the snow routine is on. + maxbas: Whether MAXBAS routing is on. + + Test scenario: + `ParameterSet` holds every trial vector to `PARAMETER_COUNTS[(snow, maxbas)]`, but + nothing held the *bounds* to it -- so a mismatch surfaced once per trial, from inside + the objective, after the whole optimisation problem had been declared and the + optimiser started. The bounds are where the configuration enters; the rule belongs + there too. + """ + coello = Calibration(Catchment("rrm", coello_rrm_date[0], coello_rrm_date[1])) + + with pytest.raises(ValueError): + coello.read_parameters_bound( + [0.0] * width, [1.0] * width, snow, maxbas=maxbas + ) + + def test_lumped_calibration( coello_rrm_date: list, lumped_meteo_data_path: str, From 8f2ea95ba7702158c4cb737373369598bc388f2e Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Thu, 10 Sep 2026 22:40:05 +0200 Subject: [PATCH 27/54] fix(calibration): copy a trial's parameters out of the optimizer's reused buffer `SimulationResults.run` is documented as "the validated inputs these arrays came from, carried as provenance". In a calibration it was not: `SpatialVarFun.Function` fills the *same* `Par3d` buffer on every trial, and `_parameter_set` wrapped that buffer by reference. Every `ParameterSet` -- and every results object reached through it -- therefore held a view of an array the next trial overwrote in place, so `results.run.parameters.values` described whichever trial happened to run last rather than the one that produced those arrays. One copy per trial, against a model run per trial, so the cost is not the point of comparison here; the alternative was a provenance field that cannot be trusted on the one path that produces thousands of result objects. --- src/hapi/calibration.py | 11 +++++++-- .../test_calibration_distributed.py | 23 +++++++++++++++++++ 2 files changed, 32 insertions(+), 2 deletions(-) diff --git a/src/hapi/calibration.py b/src/hapi/calibration.py index ec2026f4..d168d2cd 100644 --- a/src/hapi/calibration.py +++ b/src/hapi/calibration.py @@ -206,6 +206,12 @@ def _parameter_set(self, values) -> ParameterSet: `read_parameters_bound` rather than by reading a parameter file. Either source works; this picks whichever ran. + The values are copied. `SpatialVarFun` fills the *same* `Par3d` buffer on every + trial, so wrapping it by reference gave every `ParameterSet` -- and, through + `SimulationResults.run`, every set of results -- a view of an array the next trial + overwrites in place. `results.run` is documented as the inputs those arrays came + from; without the copy it described whichever trial happened to run last. + Args: values: The trial parameter array or vector. @@ -215,12 +221,13 @@ def _parameter_set(self, values) -> ParameterSet: Raises: ValueError: The trial set is not the width the configuration requires. """ + settled = np.array(values, copy=True) if self.model.parameters is not None: - return self.model.parameters.with_values(values) + return self.model.parameters.with_values(settled) bounds = self.bounds snow = bounds.snow if bounds is not None else False maxbas = bounds.maxbas if bounds is not None else False - return ParameterSet(values, snow=snow, maxbas=maxbas) + return ParameterSet(settled, snow=snow, maxbas=maxbas) def _check_before_optimising(self, **narrowing: Any) -> None: """Fail before the optimiser is built rather than on its first trial. diff --git a/tests/rrm/calibration/test_calibration_distributed.py b/tests/rrm/calibration/test_calibration_distributed.py index 818cb473..a2df16f2 100644 --- a/tests/rrm/calibration/test_calibration_distributed.py +++ b/tests/rrm/calibration/test_calibration_distributed.py @@ -282,6 +282,29 @@ def divides_by_zero(observed, gauges): "have been driven" ) + def test_a_trial_parameter_set_does_not_alias_the_optimizer_buffer( + self, gauged_calibration: Calibration + ): + """Test that a trial's parameters are copied out of the reused buffer. + + Test scenario: + `SpatialVarFun` fills the *same* `Par3d` array on every trial. Wrapping it by + reference gave every `ParameterSet` -- and, through `SimulationResults.run`, every + set of results -- a view of an array the next trial overwrites in place, so + `results.run.parameters.values` described whichever trial ran last rather than the + one that produced those arrays. + """ + coello = gauged_calibration + rows, cols = coello.model.flow_network.rows, coello.model.flow_network.cols + buffer = np.ones((rows, cols, 12)) + + settled = coello._parameter_set(buffer) + buffer[:] = 99.0 + + assert np.array_equal(settled.values, np.ones((rows, cols, 12))), ( + "the parameter set must not follow the buffer the next trial overwrites" + ) + def test_rejects_meteo_that_does_not_cover_the_grid( self, gauged_calibration: Calibration, stub_optimizer: dict, spatial_var_stub ): From 3a84f2e2622a01017cd6067eaea87d6001d5b7d6 Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Thu, 10 Sep 2026 22:40:06 +0200 Subject: [PATCH 28/54] perf(period): build the derived calendar once `SimulationPeriod` is frozen precisely so its derived values cannot drift from the inputs they come from, which also makes them safe to memoise. `date_index` rebuilt its `pd.date_range` on every read anyway, and `days` and `__len__` go through it. `DistributedRun.__post_init__` and `LumpedRun.__post_init__` each read it once per calibration trial, and `SimulationResults._step_bounds` reads it twice per call. `cached_property` is what `FlowNetwork.acc_val` and `cells_by_acc_val` already use for the same reason. --- src/hapi/period.py | 8 ++++++-- .../test_read_parameters_validation.py | 19 +++++++++++++++++++ 2 files changed, 25 insertions(+), 2 deletions(-) diff --git a/src/hapi/period.py b/src/hapi/period.py index b3db1068..affbc56f 100644 --- a/src/hapi/period.py +++ b/src/hapi/period.py @@ -16,6 +16,7 @@ import datetime as dt from dataclasses import dataclass +from functools import cached_property from typing import Literal import pandas as pd @@ -148,12 +149,15 @@ def freq(self) -> str: """str: The pandas offset alias for this resolution.""" return RESOLUTIONS[self.temporal_resolution] - @property + @cached_property def date_index(self) -> pd.DatetimeIndex: """pandas.DatetimeIndex: One entry per step, from :attr:`start` to :attr:`end`. Derived rather than stored: this is the value that used to be computed in the - constructor and could then outlive a change to the span it described. + constructor and could then outlive a change to the span it described. Cached rather + than rebuilt, which the frozen class makes safe -- the inputs it derives from cannot + change, so the cache cannot go stale. It is read once per `from_model` (so once per + calibration trial) and twice per `SimulationResults._step_bounds` call. """ return pd.date_range(self.start, self.end, freq=self.freq) diff --git a/tests/rrm/catchment/test_read_parameters_validation.py b/tests/rrm/catchment/test_read_parameters_validation.py index 640ef399..0f4b5046 100644 --- a/tests/rrm/catchment/test_read_parameters_validation.py +++ b/tests/rrm/catchment/test_read_parameters_validation.py @@ -13,6 +13,7 @@ import pytest from hapi.catchment import Catchment +from hapi.period import SimulationPeriod from hapi.rrm.hbv_bergestrom92 import HBVBergestrom92 as HBVLumped MAXBAS_BANDS = 11 @@ -85,6 +86,24 @@ def test_unknown_resolution_is_rejected_at_construction(self): with pytest.raises(ValueError, match="temporal resolutions"): Catchment("coello", "2009-01-01", "2009-01-10", temporal_resolution="15min") + def test_the_calendar_is_built_once(self): + """Test that the derived calendar is cached rather than rebuilt on every read. + + Test scenario: + The class is frozen precisely so derived values cannot drift, which makes the + calendar safe to memoise -- yet `date_index`, and `days` and `__len__` through + it, rebuilt a `pd.date_range` on each access. `from_model` reads it once per + calibration trial and `SimulationResults._step_bounds` twice per call. + """ + period = SimulationPeriod.parse("2009-01-01", "2009-12-31") + + assert period.date_index is period.date_index, ( + "the calendar cannot change on a frozen period, so it should be built once" + ) + assert len(period) == len(period.date_index), ( + "the cached index must still be what the length reports" + ) + def test_hourly_resolution_scales_the_conversion_factor(self): """Test that the hourly branch divides the daily conversion factor by 24. From 40baad49fbb98152a9a8b1a0f03d356c00925254 Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Thu, 10 Sep 2026 22:41:22 +0200 Subject: [PATCH 29/54] docs(catchment): stop promising result properties that do not exist The `Catchment` class docstring said the result arrays "are also readable under their historical names (`q_total`, `quz`, ...) as read-only properties forwarding to it". No such properties exist -- `grep "@property" src/hapi/catchment.py` finds none. That plan was written down when the results object was introduced and then deliberately dropped: the whole point of the redesign is that `results` is the only home for the arrays. The sentence was not harmless. `coello-distributed-model-run-netcdf.py` is built on it: its loop was renamed from `Qtot` to `q_total` but left pointing at the catchment, so the shipped script raised `AttributeError: 'Catchment' object has no attribute 'q_total'` for all three fields. It reads them off `Coello.results` now, and the script runs end to end. Three copies of the same claim went with it -- the `Outputs: ... [numpy attribute]` blocks in two run scripts and `tests/run/distributed_mode_run.py`, which named the arrays as attributes of the model, and the sentence in `docs/examples/distributed-model-calib.md` saying the results "will be stored as attributes in the Catchment object". --- docs/examples/distributed-model-calib.md | 2 +- .../run/coello-distributed-model-run-maxbas.py | 15 +++++++-------- .../run/coello-distributed-model-run-netcdf.py | 17 ++++++++--------- src/hapi/catchment.py | 5 +++-- tests/run/distributed_mode_run.py | 15 +++++++-------- 5 files changed, 26 insertions(+), 28 deletions(-) diff --git a/docs/examples/distributed-model-calib.md b/docs/examples/distributed-model-calib.md index 22592b27..7cc31c9a 100644 --- a/docs/examples/distributed-model-calib.md +++ b/docs/examples/distributed-model-calib.md @@ -106,7 +106,7 @@ Coello.read_discharge_gauges(GaugesPath, column='id', fmt="%Y-%m-%d") from hapi.run import Run Run.run_distributed(Coello) ``` -- the result of the simulation will be stored as attributes in the Catchment object as follow +- the result of the simulation is returned, and also assigned to `Coello.results` as follow ```python """ diff --git a/examples/hydrological-model/coello/run/coello-distributed-model-run-maxbas.py b/examples/hydrological-model/coello/run/coello-distributed-model-run-maxbas.py index 04b5cb46..1fa56dc7 100644 --- a/examples/hydrological-model/coello/run/coello-distributed-model-run-maxbas.py +++ b/examples/hydrological-model/coello/run/coello-distributed-model-run-maxbas.py @@ -24,21 +24,20 @@ # %% Run the model """ -Outputs: - ---------- - 1-state_variables: [numpy attribute] +Outputs, all on `Coello.results` (a `SimulationResults`) once the run returns: + 1-state_variables: 4D array (rows,cols,time,states) states are [sp,wc,sm,uz,lv] - 2-qlz: [numpy attribute] + 2-qlz: 3D array of the lower zone discharge - 3-quz: [numpy attribute] + 3-quz: 3D array of the upper zone discharge - 4-qout: [numpy attribute] + 4-qout: 1D timeseries of discharge at the outlet of the catchment of unit m3/sec - 5-quz_routed: [numpy attribute] + 5-quz_routed: 3D array of the upper zone discharge accumulated and routed at each time step - 6-qlz_translated: [numpy attribute] + 6-qlz_translated: 3D array of the lower zone discharge translated at each time step """ Run.run_maxbas(Coello) diff --git a/examples/hydrological-model/coello/run/coello-distributed-model-run-netcdf.py b/examples/hydrological-model/coello/run/coello-distributed-model-run-netcdf.py index 70940945..432cb94d 100644 --- a/examples/hydrological-model/coello/run/coello-distributed-model-run-netcdf.py +++ b/examples/hydrological-model/coello/run/coello-distributed-model-run-netcdf.py @@ -44,21 +44,20 @@ # %% Run the model """ -Outputs: - ---------- - 1-state_variables: [numpy attribute] +Outputs, all on `Coello.results` (a `SimulationResults`) once the run returns: + 1-state_variables: 4D array (rows,cols,time,states) states are [sp,wc,sm,uz,lv] - 2-qlz: [numpy attribute] + 2-qlz: 3D array of the lower zone discharge - 3-quz: [numpy attribute] + 3-quz: 3D array of the upper zone discharge - 4-qout: [numpy attribute] + 4-qout: 1D timeseries of discharge at the outlet of the catchment of unit m3/sec - 5-quz_routed: [numpy attribute] + 5-quz_routed: 3D array of the upper zone discharge accumulated and routed at each time step - 6-qlz_translated: [numpy attribute] + 6-qlz_translated: 3D array of the lower zone discharge translated at each time step """ Run.run_distributed(Coello) @@ -66,7 +65,7 @@ # %% Routed fields cover the grid, finite inside the catchment inside = ~np.isnan(Coello.flow_network.flow_acc_arr) for field_name in ("q_total", "quz_routed", "qlz_translated"): - field = getattr(Coello, field_name) + field = getattr(Coello.results, field_name) print( f"{field_name:15s} shape {field.shape}, finite inside: {np.isfinite(field[inside]).all()}" ) diff --git a/src/hapi/catchment.py b/src/hapi/catchment.py index 4a511213..954e62be 100644 --- a/src/hapi/catchment.py +++ b/src/hapi/catchment.py @@ -228,8 +228,9 @@ class Catchment: needs as a protocol, which this class satisfies structurally; neither class inherits from the other. - A run assigns its output to :attr:`results`. The result arrays are also readable under - their historical names (`q_total`, `quz`, ...) as read-only properties forwarding to it. + A run assigns its output to :attr:`results`, and that is the only place the arrays + live: read them as `model.results.q_total`, `model.results.quz` and so on. This class + carries no result attributes of its own and no forwarding properties. """ def __init__( diff --git a/tests/run/distributed_mode_run.py b/tests/run/distributed_mode_run.py index 0f2ccd94..d60cfdf1 100644 --- a/tests/run/distributed_mode_run.py +++ b/tests/run/distributed_mode_run.py @@ -40,21 +40,20 @@ Coello.read_discharge_gauges(GaugesPath, column="id", fmt="%Y-%m-%d") # %% Run the model """ -Outputs: - ---------- - 1-state_variables: [numpy attribute] +Outputs, all on `Coello.results` (a `SimulationResults`) once the run returns: + 1-state_variables: 4D array (rows,cols,time,states) states are [sp,wc,sm,uz,lv] - 2-qlz: [numpy attribute] + 2-qlz: 3D array of the lower zone discharge - 3-quz: [numpy attribute] + 3-quz: 3D array of the upper zone discharge - 4-qout: [numpy attribute] + 4-qout: 1D timeseries of discharge at the outlet of the catchment of unit m3/sec - 5-quz_routed: [numpy attribute] + 5-quz_routed: 3D array of the upper zone discharge accumulated and routed at each time step - 6-qlz_translated: [numpy attribute] + 6-qlz_translated: 3D array of the lower zone discharge translated at each time step """ Run.run_distributed(Coello) From 8ee0ac2bf9f1b6ce0eb168fe41c7e4aebc3815a3 Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Thu, 10 Sep 2026 22:43:37 +0200 Subject: [PATCH 30/54] fix(examples): finish the Calibration is-a to has-a migration in the shipped scripts `Calibration` stopped being a `Catchment` and started holding one. The scripts were migrated by prefixing the reader calls with `.model` -- and nothing else, so every one of them failed on its first migrated line. Five calibration examples and two `tests/` scripts reached through the wrapper in four more places: * `Run.run_lumped(Coello, ...)` passed the `Calibration` where a catchment is required: `AttributeError: 'Calibration' object has no attribute 'period'`. * `Coello.Qsim["q"]` -- `Run.run_lumped` puts the frame on the *model*; `Calibration.Qsim` is `None`, or a bare array from `extract_discharge`, never a frame with a `"q"` column. * `Coello.name`, used to build the output filenames in the deap scripts. `Calibration` has no `name`. * `Coello.model.parameters = ` handed the engine a list where a `ParameterSet` is required, so `run.parameters.snow` raised. The `...parameters.values = ...` variants were worse: `ParameterSet` is frozen, so they raised `FrozenInstanceError`. They build the set from the same `(snow, maxbas)` pair the bounds were read with. Two more that came with the same change: `coello-hrus-...-muskingum.py` passed `model.parameters` -- now a `ParameterSet` -- to `SpatialVarFun.Function`, which wants the flat vector, so it passes `Coello.best_parameters`; and `tests/sensitivity_analysis.py` called `read_parameters_bound` on a plain `Catchment`, which no longer has it, so it samples between the two bound lists it had already read. Verified by running the post-calibration sequence the scripts perform -- install the optimiser's vector, run, score against the gauges -- end to end. --- ...stributed-model-calibratation-muskingum.py | 5 ++- ...libration-deap-multiobjective-NSE-NSEHF.py | 31 +++++++++++-------- ...alibration-deap-multiobjective-NSE-RMSE.py | 31 +++++++++++-------- .../coello-lumped-model-calibration-deap.py | 29 ++++++++++------- .../coello-lumped-model-calibration.py | 19 +++++++----- tests/calibration/lumped_calibration.py | 19 +++++++----- tests/sensitivity_analysis.py | 11 ++++--- 7 files changed, 87 insertions(+), 58 deletions(-) diff --git a/examples/hydrological-model/coello/calibration/coello-hrus-distributed-model-calibratation-muskingum.py b/examples/hydrological-model/coello/calibration/coello-hrus-distributed-model-calibratation-muskingum.py index 60627fc0..c8aa7cd8 100644 --- a/examples/hydrological-model/coello/calibration/coello-hrus-distributed-model-calibratation-muskingum.py +++ b/examples/hydrological-model/coello/calibration/coello-hrus-distributed-model-calibratation-muskingum.py @@ -139,7 +139,10 @@ def objective_function(q_obs, coordinates): # Qout, q_uz_routed, q_lz_trans, # %% run calibration cal_parameters = Coello.run_calibration(SpatialVarFun, OptimizationArgs, print_error=0) # %% convert parameters to rasters +# `best_parameters` is the flat vector the optimiser produced, which is what this +# function maps onto the grid. `model.parameters` is a `ParameterSet` -- a different +# shape describing a different thing. SpatialVarFun.Function( - Coello.model.parameters, kub=SpatialVarFun.Kub, klb=SpatialVarFun.Klb + Coello.best_parameters, kub=SpatialVarFun.Kub, klb=SpatialVarFun.Klb ) SpatialVarFun.save_parameters(SaveTo) diff --git a/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-NSEHF.py b/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-NSEHF.py index b7465624..976deba8 100644 --- a/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-NSEHF.py +++ b/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-NSEHF.py @@ -17,6 +17,7 @@ from hapi.calibration import Calibration from hapi.catchment import Catchment +from hapi.conceptual import ParameterSet from hapi.routing import Routing from hapi.rrm.hbv_bergestrom92 import HBVBergestrom92 as HBVLumped from hapi.run import Run @@ -95,11 +96,13 @@ def initializer(): def objfn(individual): # Coello.model.read_parameters(Parameterpath, Snow) - Coello.model.parameters = individual - Run.run_lumped(Coello, Route, RoutingFn) + Coello.model.parameters = ParameterSet( + individual, snow=Coello.bounds.snow, maxbas=Coello.bounds.maxbas +) + Run.run_lumped(Coello.model, Route, RoutingFn) # [Coello.model.QGauges.columns[-1]] - NSE = metrics.nse_hf(Coello.model.QGauges, Coello.Qsim, *Coello.OFArgs) - NSEHF = metrics.nse_hf(Coello.model.QGauges, Coello.Qsim, *Coello.OFArgs) + NSE = metrics.nse_hf(Coello.model.QGauges, Coello.model.Qsim, *Coello.OFArgs) + NSEHF = metrics.nse_hf(Coello.model.QGauges, Coello.model.Qsim, *Coello.OFArgs) return NSE, NSEHF @@ -148,11 +151,13 @@ def distance(individual): print("Best individual is %s, %s" % (best_ind, best_ind.fitness.values)) # %% Run the Model -Coello.model.parameters = best_ind +Coello.model.parameters = ParameterSet( + best_ind, snow=Coello.bounds.snow, maxbas=Coello.bounds.maxbas +) # [0.7686518278956287, 144.35510831203874, 1.9922719933560913, 0.1439126168555068, 0.9474744708723734, # 0.749219030317463, 0.8074091462437563, 0.07289588281400794, 68.83482640397304, 5.123384184968337, # 1.9922719933560913] -Run.run_lumped(Coello, Route, RoutingFn) +Run.run_lumped(Coello.model, Route, RoutingFn) ### Calculate Performance Criteria @@ -160,11 +165,11 @@ def distance(individual): Qobs = Coello.model.QGauges[Coello.model.QGauges.columns[0]] -scores["RMSE"] = metrics.rmse(Qobs, Coello.Qsim["q"]) -scores["NSE"] = metrics.nse(Qobs, Coello.Qsim["q"]) -scores["NSEhf"] = metrics.nse_hf(Qobs, Coello.Qsim["q"]) -scores["KGE"] = metrics.kge(Qobs, Coello.Qsim["q"]) -scores["WB"] = metrics.wb(Qobs, Coello.Qsim["q"]) +scores["RMSE"] = metrics.rmse(Qobs, Coello.model.Qsim["q"]) +scores["NSE"] = metrics.nse(Qobs, Coello.model.Qsim["q"]) +scores["NSEhf"] = metrics.nse_hf(Qobs, Coello.model.Qsim["q"]) +scores["KGE"] = metrics.kge(Qobs, Coello.model.Qsim["q"]) +scores["WB"] = metrics.wb(Qobs, Coello.model.Qsim["q"]) print("RMSE= " + str(round(scores["RMSE"], 2))) print("NSE= " + str(round(scores["NSE"], 2))) @@ -182,7 +187,7 @@ def distance(individual): ParPath = ( Path - + f"{Coello.name}-lumped-parameters-multi-obj" + + f"{Coello.model.name}-lumped-parameters-multi-obj" + str(dt.datetime.now())[0:10] + ".txt" ) @@ -197,7 +202,7 @@ def distance(individual): Path = ( Path - + f"{Coello.name}-results-lumped-model-multi-obj" + + f"{Coello.model.name}-results-lumped-model-multi-obj" + str(dt.datetime.now())[0:10] + ".txt" ) diff --git a/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-RMSE.py b/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-RMSE.py index c33dc6ba..eaa7f494 100644 --- a/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-RMSE.py +++ b/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-RMSE.py @@ -18,6 +18,7 @@ from hapi.calibration import Calibration from hapi.catchment import Catchment +from hapi.conceptual import ParameterSet from hapi.routing import Routing from hapi.rrm.hbv_bergestrom92 import HBVBergestrom92 as HBVLumped from hapi.run import Run @@ -97,11 +98,13 @@ def initializer(): def objfn(individual): # Coello.model.read_parameters(Parameterpath, Snow) - Coello.model.parameters = individual - Run.run_lumped(Coello, Route, RoutingFn) + Coello.model.parameters = ParameterSet( + individual, snow=Coello.bounds.snow, maxbas=Coello.bounds.maxbas +) + Run.run_lumped(Coello.model, Route, RoutingFn) # [Coello.model.QGauges.columns[-1]] - NSE = metrics.nse_hf(Coello.model.QGauges, Coello.Qsim, *Coello.OFArgs) - RMSE = metrics.rmse(Coello.model.QGauges, Coello.Qsim, *Coello.OFArgs) + NSE = metrics.nse_hf(Coello.model.QGauges, Coello.model.Qsim, *Coello.OFArgs) + RMSE = metrics.rmse(Coello.model.QGauges, Coello.model.Qsim, *Coello.OFArgs) return NSE, RMSE @@ -150,11 +153,13 @@ def distance(individual): print("Best individual is %s, %s" % (best_ind, best_ind.fitness.values)) # %% Run the Model -Coello.model.parameters = best_ind +Coello.model.parameters = ParameterSet( + best_ind, snow=Coello.bounds.snow, maxbas=Coello.bounds.maxbas +) # [0.7686518278956287, 144.35510831203874, 1.9922719933560913, 0.1439126168555068, 0.9474744708723734, # 0.749219030317463, 0.8074091462437563, 0.07289588281400794, 68.83482640397304, 5.123384184968337, # 1.9922719933560913] -Run.run_lumped(Coello, Route, RoutingFn) +Run.run_lumped(Coello.model, Route, RoutingFn) ### Calculate Performance Criteria @@ -162,11 +167,11 @@ def distance(individual): Qobs = Coello.model.QGauges[Coello.model.QGauges.columns[0]] -scores["RMSE"] = metrics.rmse(Qobs, Coello.Qsim["q"]) -scores["NSE"] = metrics.nse(Qobs, Coello.Qsim["q"]) -scores["NSEhf"] = metrics.nse_hf(Qobs, Coello.Qsim["q"]) -scores["KGE"] = metrics.kge(Qobs, Coello.Qsim["q"]) -scores["WB"] = metrics.wb(Qobs, Coello.Qsim["q"]) +scores["RMSE"] = metrics.rmse(Qobs, Coello.model.Qsim["q"]) +scores["NSE"] = metrics.nse(Qobs, Coello.model.Qsim["q"]) +scores["NSEhf"] = metrics.nse_hf(Qobs, Coello.model.Qsim["q"]) +scores["KGE"] = metrics.kge(Qobs, Coello.model.Qsim["q"]) +scores["WB"] = metrics.wb(Qobs, Coello.model.Qsim["q"]) print("RMSE= " + str(round(scores["RMSE"], 2))) print("NSE= " + str(round(scores["NSE"], 2))) @@ -185,7 +190,7 @@ def distance(individual): ParPath = ( Path - + f"{Coello.name}-lumped-parameters-multi-obj" + + f"{Coello.model.name}-lumped-parameters-multi-obj" + str(dt.datetime.now())[0:10] + ".txt" ) @@ -200,7 +205,7 @@ def distance(individual): Path = ( Path - + f"{Coello.name}-results-lumped-model-multi-obj" + + f"{Coello.model.name}-results-lumped-model-multi-obj" + str(dt.datetime.now())[0:10] + ".txt" ) diff --git a/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap.py b/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap.py index 47263942..05b60b6c 100644 --- a/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap.py +++ b/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap.py @@ -16,6 +16,7 @@ from hapi.calibration import Calibration from hapi.catchment import Catchment +from hapi.conceptual import ParameterSet from hapi.routing import Routing from hapi.rrm.hbv_bergestrom92 import HBVBergestrom92 as HBVLumped from hapi.run import Run @@ -93,10 +94,12 @@ def initializer(): def objfn(individual): # Coello.model.read_parameters(Parameterpath, Snow) - Coello.model.parameters = individual - Run.run_lumped(Coello, Route, RoutingFn) + Coello.model.parameters = ParameterSet( + individual, snow=Coello.bounds.snow, maxbas=Coello.bounds.maxbas +) + Run.run_lumped(Coello.model, Route, RoutingFn) # [Coello.model.QGauges.columns[-1]] - error = PC.NSEHF(Coello.model.QGauges, Coello.Qsim, *Coello.OFArgs) + error = PC.NSEHF(Coello.model.QGauges, Coello.model.Qsim, *Coello.OFArgs) return (error,) @@ -145,11 +148,13 @@ def distance(individual): print("Best individual is %s, %s" % (best_ind, best_ind.fitness.values)) # %% Run the Model -Coello.model.parameters = best_ind +Coello.model.parameters = ParameterSet( + best_ind, snow=Coello.bounds.snow, maxbas=Coello.bounds.maxbas +) # [0.7686518278956287, 144.35510831203874, 1.9922719933560913, 0.1439126168555068, 0.9474744708723734, # 0.749219030317463, 0.8074091462437563, 0.07289588281400794, 68.83482640397304, 5.123384184968337, # 1.9922719933560913] -Run.run_lumped(Coello, Route, RoutingFn) +Run.run_lumped(Coello.model, Route, RoutingFn) ### Calculate Performance Criteria @@ -157,11 +162,11 @@ def distance(individual): Qobs = Coello.model.QGauges[Coello.model.QGauges.columns[0]] -metrics["RMSE"] = PC.RMSE(Qobs, Coello.Qsim["q"]) -metrics["NSE"] = PC.NSE(Qobs, Coello.Qsim["q"]) -metrics["NSEhf"] = PC.NSEHF(Qobs, Coello.Qsim["q"]) -metrics["KGE"] = PC.KGE(Qobs, Coello.Qsim["q"]) -metrics["WB"] = PC.WB(Qobs, Coello.Qsim["q"]) +metrics["RMSE"] = PC.RMSE(Qobs, Coello.model.Qsim["q"]) +metrics["NSE"] = PC.NSE(Qobs, Coello.model.Qsim["q"]) +metrics["NSEhf"] = PC.NSEHF(Qobs, Coello.model.Qsim["q"]) +metrics["KGE"] = PC.KGE(Qobs, Coello.model.Qsim["q"]) +metrics["WB"] = PC.WB(Qobs, Coello.model.Qsim["q"]) print("RMSE= " + str(round(metrics["RMSE"], 2))) print("NSE= " + str(round(metrics["NSE"], 2))) @@ -179,7 +184,7 @@ def distance(individual): # %% Save the Parameters ParPath = ( - Path + f"{Coello.name}-lumped-parameters" + str(dt.datetime.now())[0:10] + ".txt" + Path + f"{Coello.model.name}-lumped-parameters" + str(dt.datetime.now())[0:10] + ".txt" ) parameters = pd.DataFrame(index=parnames) # parameters['values'] = cal_parameters[1] @@ -191,6 +196,6 @@ def distance(individual): EndDate = "2010-04-20" Path = ( - Path + f"{Coello.name}-results-lumped-model" + str(dt.datetime.now())[0:10] + ".txt" + Path + f"{Coello.model.name}-results-lumped-model" + str(dt.datetime.now())[0:10] + ".txt" ) Coello.model.results.save(result=5, start=StartDate, end=EndDate, path=Path) diff --git a/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration.py b/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration.py index 8ca38b41..afb9e9ca 100644 --- a/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration.py +++ b/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration.py @@ -9,6 +9,7 @@ from hapi.calibration import Calibration from hapi.catchment import Catchment +from hapi.conceptual import ParameterSet from hapi.routing import Routing from hapi.rrm.hbv_bergestrom92 import HBVBergestrom92 as HBVLumped from hapi.run import Run @@ -107,8 +108,12 @@ print("Time = " + str(round(cal_parameters[2]["time"] / 60, 2)) + " min") # %% Run the Model -Coello.model.parameters = cal_parameters[1] -Run.run_lumped(Coello, Route, RoutingFn) +# The optimiser hands back a flat vector; the model holds a checked `ParameterSet`, +# whose width rule comes from the same (snow, maxbas) pair the bounds were read with. +Coello.model.parameters = ParameterSet( + cal_parameters[1], snow=Coello.bounds.snow, maxbas=Coello.bounds.maxbas +) +Run.run_lumped(Coello.model, Route, RoutingFn) ### Calculate Performance Criteria @@ -116,11 +121,11 @@ Qobs = Coello.model.QGauges[Coello.model.QGauges.columns[0]] -scores["RMSE"] = metrics.rmse(Qobs, Coello.Qsim["q"]) -scores["NSE"] = metrics.nse(Qobs, Coello.Qsim["q"]) -scores["NSEhf"] = metrics.nse_hf(Qobs, Coello.Qsim["q"]) -scores["KGE"] = metrics.kge(Qobs, Coello.Qsim["q"]) -scores["WB"] = metrics.wb(Qobs, Coello.Qsim["q"]) +scores["RMSE"] = metrics.rmse(Qobs, Coello.model.Qsim["q"]) +scores["NSE"] = metrics.nse(Qobs, Coello.model.Qsim["q"]) +scores["NSEhf"] = metrics.nse_hf(Qobs, Coello.model.Qsim["q"]) +scores["KGE"] = metrics.kge(Qobs, Coello.model.Qsim["q"]) +scores["WB"] = metrics.wb(Qobs, Coello.model.Qsim["q"]) print("RMSE= " + str(round(scores["RMSE"], 2))) print("NSE= " + str(round(scores["NSE"], 2))) diff --git a/tests/calibration/lumped_calibration.py b/tests/calibration/lumped_calibration.py index c3b01318..6c7f341c 100644 --- a/tests/calibration/lumped_calibration.py +++ b/tests/calibration/lumped_calibration.py @@ -6,6 +6,7 @@ from hapi.calibration import Calibration from hapi.catchment import Catchment +from hapi.conceptual import ParameterSet from hapi.routing import Routing from hapi.rrm.hbv_bergestrom92 import HBVBergestrom92 as HBVLumped from hapi.run import Run @@ -96,18 +97,22 @@ print("Parameters are " + str(cal_parameters[1])) print("Time = " + str(round(cal_parameters[2]["time"] / 60, 2)) + " min") # %% run the model -Coello.model.parameters.values = cal_parameters[1] -Run.run_lumped(Coello, Route, routing_fn) +# `ParameterSet` is frozen, so the values cannot be assigned into it; build the set +# the optimiser's vector describes. +Coello.model.parameters = ParameterSet( + cal_parameters[1], snow=Coello.bounds.snow, maxbas=Coello.bounds.maxbas +) +Run.run_lumped(Coello.model, Route, routing_fn) # %% calculate performance criteria scores = dict() Qobs = Coello.model.QGauges[Coello.model.QGauges.columns[0]] -scores["RMSE"] = metrics.rmse(Qobs, Coello.Qsim["q"]) -scores["NSE"] = metrics.nse(Qobs, Coello.Qsim["q"]) -scores["NSEhf"] = metrics.nse_hf(Qobs, Coello.Qsim["q"]) -scores["KGE"] = metrics.kge(Qobs, Coello.Qsim["q"]) -scores["WB"] = metrics.wb(Qobs, Coello.Qsim["q"]) +scores["RMSE"] = metrics.rmse(Qobs, Coello.model.Qsim["q"]) +scores["NSE"] = metrics.nse(Qobs, Coello.model.Qsim["q"]) +scores["NSEhf"] = metrics.nse_hf(Qobs, Coello.model.Qsim["q"]) +scores["KGE"] = metrics.kge(Qobs, Coello.model.Qsim["q"]) +scores["WB"] = metrics.wb(Qobs, Coello.model.Qsim["q"]) print("RMSE= " + str(round(scores["RMSE"], 2))) print("NSE= " + str(round(scores["NSE"], 2))) diff --git a/tests/sensitivity_analysis.py b/tests/sensitivity_analysis.py index f3700f6b..a2aa833d 100644 --- a/tests/sensitivity_analysis.py +++ b/tests/sensitivity_analysis.py @@ -40,7 +40,8 @@ UB = UB[1].tolist() LB = pd.read_csv(Path + "/UB-1-Muskinguk.txt", index_col=0, header=None) LB = LB[1].tolist() -Coello.read_parameters_bound(UB, LB, Snow) +# The bounds moved onto `Calibration` with the is-a -> has-a change, and this script +# only samples between them -- so it uses the two lists it just read. # %% # observed flow @@ -111,7 +112,7 @@ # For Type 1 def WrapperType1(Randpar, Route, routing_fn, Qobs): - Coello.parameters.values = Randpar + Coello.parameters = Coello.parameters.with_values(Randpar) Run.run_lumped(Coello, Route, routing_fn) rmse = metrics.rmse(Qobs, Coello.Qsim["q"]) @@ -120,7 +121,7 @@ def WrapperType1(Randpar, Route, routing_fn, Qobs): # For Type 2 def WrapperType2(Randpar, Route, routing_fn, Qobs): - Coello.parameters.values = Randpar + Coello.parameters = Coello.parameters.with_values(Randpar) Run.run_lumped(Coello, Route, routing_fn) rmse = metrics.rmse(Qobs, Coello.Qsim["q"]) @@ -137,8 +138,8 @@ def WrapperType2(Randpar, Route, routing_fn, Qobs): Sen = SA( parameters, - Coello.bounds.lower, - Coello.bounds.upper, + LB, + UB, fn, n_values=5, return_values=Type, From 7a8e2d9c19754086d2ad019347640485e0e2e6f1 Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Thu, 10 Sep 2026 22:45:06 +0200 Subject: [PATCH 31/54] docs: correct the pages that still describe removed calibration and run APIs Three pages were left describing APIs this branch removed, each in a way a reader following along cannot get past. `run-configuration.md` had the routing function passed as `Run.run_lumped`'s `Route` flag. A callable is truthy, so `Route != 0` while `routing_fn` stays `None`, and the call raises. The line was edited here (`runLumped` -> `run_lumped`) without the argument order being noticed. It also still named `Run.RunHapiwithLake`, whose neighbours in the same paragraph had been renamed, and promised that `Calibration.from_yaml` "works -- it takes the same constructor arguments". It does not exist: `Calibration` is no longer a `Catchment` subclass, so it inherits nothing. The page shows `Calibration(Catchment.from_yaml(path))`. `lumped-model-calibration.md` had only its last line updated (`lumpedCalibration` -> `calibrate_lumped`); everything above still built `Calibration(name, start, end)` and called the readers on it. Two smaller errors went with it: `Maxbas=` for a parameter spelled `maxbas`, and `Snow = 0` where `read_parameters_bound` requires a bool. Its imports were stale too -- `statista.metrics` and a module used as a class. `test_config.py`'s "builder returns the class it was called on" test had been reduced to a single-element parametrize while its docstring still explained that "`Calibration` extends `Catchment` with the same constructor signature". It states what it actually guards now: `from_yaml` constructs `cls`, so the classmethod stays inheritable, which is the only reason the test is worth keeping. --- docs/examples/lumped-model-calibration.md | 21 +++++++++++-------- docs/examples/run-configuration.md | 16 +++++++++------ tests/rrm/catchment/test_config.py | 25 ++++++++++++++--------- 3 files changed, 38 insertions(+), 24 deletions(-) diff --git a/docs/examples/lumped-model-calibration.md b/docs/examples/lumped-model-calibration.md index f3858b13..40cddf3d 100644 --- a/docs/examples/lumped-model-calibration.md +++ b/docs/examples/lumped-model-calibration.md @@ -6,11 +6,13 @@ To calibrate the HBV lumped model inside Hapi you need to follow the same steps ```python import pandas as pd import datetime as dt - import hapi.rrm.hbv_bergestrom92 as HBVLumped + from hapi.rrm.hbv_bergestrom92 import HBVBergestrom92 as HBVLumped from hapi.calibration import Calibration + from hapi.catchment import Catchment + from hapi.conceptual import ParameterSet from hapi.routing import Routing from hapi.run import Run - import statista.metrics as metrics + import statista.descriptors as metrics Parameterpath = Comp + "/data/lumped/Coello_Lumped2021-03-08_muskingum.txt" MeteoDataPath = Comp + "/data/lumped/meteo_data-MSWEP.csv" @@ -20,8 +22,10 @@ To calibrate the HBV lumped model inside Hapi you need to follow the same steps end = "2011-12-31" name = "Coello" - Coello = Calibration(name, start, end) - Coello.read_lumped_inputs(MeteoDataPath) + # `Calibration` holds a catchment rather than being one, so the model is built first + # and the readers are called on `Coello.model`. + Coello = Calibration(Catchment(name, start, end)) + Coello.model.read_lumped_inputs(MeteoDataPath) # catchment area @@ -30,8 +34,8 @@ To calibrate the HBV lumped model inside Hapi you need to follow the same steps # [Snow pack, Soil moisture, Upper zone, Lower Zone, Water content] InitialCond = [0,10,10,10,0] # no snow subroutine - Snow = 0 - Coello.read_lumped_model(HBVLumped, AreaCoeff, InitialCond) + Snow = False + Coello.model.read_lumped_model(HBVLumped, AreaCoeff, InitialCond) # Calibration boundaries UB = pd.read_csv(Path + "/lumped/UB-3.txt", index_col = 0, header = None) @@ -41,7 +45,8 @@ To calibrate the HBV lumped model inside Hapi you need to follow the same steps LB = LB[1].tolist() Maxbas = True - Coello.read_parameters_bound(UB, LB, Snow, Maxbas=Maxbas) + # `maxbas` is lower case, and `snow` has to be a bool -- `0` is refused by name. + Coello.read_parameters_bound(UB, LB, Snow, maxbas=Maxbas) parameters = [] # Routing @@ -52,7 +57,7 @@ To calibrate the HBV lumped model inside Hapi you need to follow the same steps ### Objective function # outlet discharge - Coello.read_discharge_gauges(Path+"Qout_c.csv", fmt="%Y-%m-%d") + Coello.model.read_discharge_gauges(Path+"Qout_c.csv", fmt="%Y-%m-%d") OF_args=[] OF=metrics.rmse diff --git a/docs/examples/run-configuration.md b/docs/examples/run-configuration.md index a4de0628..ae3a9fa4 100644 --- a/docs/examples/run-configuration.md +++ b/docs/examples/run-configuration.md @@ -11,10 +11,13 @@ same order: ```python from hapi.catchment import Catchment +from hapi.routing import Routing from hapi.run import Run Coello = Catchment.from_yaml("coello-lumped-model-run.yaml") -Run.run_lumped(Coello, Routing.triangular_routing_1) +# `Route` is the flag; the routing function is the third argument. Passing the function +# as the flag routes with nothing, because a callable is truthy. +Run.run_lumped(Coello, 1, Routing.triangular_routing_1) ``` The four shipped examples under `examples/hydrological-model/coello/run/` are each a pair — a @@ -157,11 +160,12 @@ Coello.results.save( ## Out of scope The schema describes a `Catchment` run. It carries no field for a lake record, a river geometry, -or a flow-path-length raster, so lake-aware runs (`Run.RunHapiwithLake`), the flood model -(`Run.run_flood`) and `route_maxbas_by_path_length` are still assembled in Python. +or a flow-path-length raster, so lake-aware runs (`Run.run_distributed_with_lake`), the flood +model (`Run.run_flood`) and `route_maxbas_by_path_length` are still assembled in Python. -`Calibration.from_yaml` works — it takes the same constructor arguments — and gives back a -`Calibration` to call the calibration methods on. `Run.from_yaml` does not: `Run` holds entry -points called on a model built elsewhere, so it refuses and says so. +`Calibration.from_yaml` does not exist: `Calibration` is no longer a `Catchment` subclass, so it +inherits nothing. Build the model from the file and hand it over — +`Calibration(Catchment.from_yaml(path))`. `Run.from_yaml` does not exist either: `Run` holds +entry points called on a model built elsewhere. The full field-by-field reference is on the [Config API page](../api/config.md). diff --git a/tests/rrm/catchment/test_config.py b/tests/rrm/catchment/test_config.py index 228c26a2..c30ed170 100644 --- a/tests/rrm/catchment/test_config.py +++ b/tests/rrm/catchment/test_config.py @@ -19,6 +19,8 @@ import os from pathlib import Path +import inspect + import pytest import yaml from pydantic import ValidationError @@ -1478,26 +1480,29 @@ def test_an_unregistered_model_class_is_refused_before_the_readers_run( with pytest.raises(ValueError, match="not.*registered"): Catchment.from_yaml(path) - @pytest.mark.parametrize("cls", [Catchment]) def test_the_builder_returns_the_class_it_was_called_on( - self, distributed_mapping, tmp_path, cls + self, distributed_mapping, tmp_path ): - """Test that a subclass taking the same constructor arguments builds its own type. + """Test that the classmethod builds its own type rather than a hard-coded one. Args: distributed_mapping: A complete distributed configuration. tmp_path: pytest temporary directory. - cls: The class the classmethod is called on. Test scenario: - `Calibration` extends `Catchment` with the same constructor signature, so - `Calibration.from_yaml(...)` should hand back a `Calibration` the calibration - methods can be called on, not a bare `Catchment`. + `from_yaml` returns `cls(...)`, not `Catchment(...)`. Nothing in the package + subclasses `Catchment` any more -- `Calibration` holds one instead, which is why + this test's parametrize had shrunk to a single element -- so the guard is what + keeps the classmethod inheritable for anyone downstream who does subclass it. """ - model = cls.from_yaml(write_yaml(distributed_mapping, tmp_path)) + model = Catchment.from_yaml(write_yaml(distributed_mapping, tmp_path)) - assert isinstance(model, cls), ( - f"expected a {cls.__name__}, got {type(model).__name__}" + assert type(model) is Catchment, ( + f"expected a Catchment, got {type(model).__name__}" + ) + assert "cls(" in inspect.getsource(Catchment.from_yaml), ( + "from_yaml must construct `cls`, not a hard-coded Catchment, or a subclass " + "gets the wrong type back" ) def test_run_has_no_constructor_and_nothing_to_build_from_a_configuration(self): From 415438f5da96aec443c585be1941e7a90e18516a Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Thu, 10 Sep 2026 22:47:11 +0200 Subject: [PATCH 32/54] docs(api): document the four new public modules, and drop the constants they replaced `docs/api/results.md` was the only page this branch added, but `SimulationPeriod`, `DistributedRun`, `LumpedRun`, `ParameterSet`, `ConceptualModelSetup`, `ParameterBounds`, `CatchmentLike` and `SpatialDistribution` are all public now -- `Catchment.period`, `.parameters` and `.model_setup` are typed with them, and docstrings across the package carry dozens of `hapi.runs.DistributedRun` style cross-references that resolved to no page at all. Four pages, each opening with why the object exists rather than only what it holds: the builder/finished split for `runs`, the six-attributes-describing-one-thing story for `period`, the per-object invariants for `conceptual`, and dependency inversion for `protocols`. `CONVERSION_FACTOR` and `PARAMETER_COUNTS` go with them. Both were dead in `catchment.py` -- only the definitions were left -- but they were still public module-level names, and a second source of truth for the two rules the refactor was consolidating: the mm-to-m3/s factor, which now lives in `period`, and the `(snow, maxbas)` count table, which lives in `conceptual`. The next person to change one would not have thought to change both. `mkdocs build --strict` stays clean. --- docs/api/conceptual.md | 29 +++++++++++++++++++++++++++++ docs/api/period.md | 21 +++++++++++++++++++++ docs/api/protocols.md | 23 +++++++++++++++++++++++ docs/api/runs.md | 26 ++++++++++++++++++++++++++ mkdocs.yml | 4 ++++ src/hapi/catchment.py | 9 --------- 6 files changed, 103 insertions(+), 9 deletions(-) create mode 100644 docs/api/conceptual.md create mode 100644 docs/api/period.md create mode 100644 docs/api/protocols.md create mode 100644 docs/api/runs.md diff --git a/docs/api/conceptual.md b/docs/api/conceptual.md new file mode 100644 index 00000000..f92e5327 --- /dev/null +++ b/docs/api/conceptual.md @@ -0,0 +1,29 @@ +# Conceptual model inputs + +The conceptual model's inputs used to be six loose attributes on `Catchment` whose rules were +enforced nowhere in particular. They are three value objects now, each checking its own invariant +in `__post_init__`, so a bad combination is refused where it is made rather than several frames +into a run. + +- `ParameterSet` — the parameter values plus the `(snow, maxbas)` pair that fixes their width. + Every route to a parameter set goes through the same width rule, including the per-trial + replacements a calibration makes. It is frozen; use `with_values` to derive a new set from an + optimiser's vector. +- `ConceptualModelSetup` — the model, the catchment area, the initial condition and the initial + discharge, as `read_lumped_model` produces them. +- `ParameterBounds` — the calibration's search space, held to the same width rule as the trial + vectors it bounds. + +## ParameterSet +::: hapi.conceptual.ParameterSet + +## ConceptualModelSetup +::: hapi.conceptual.ConceptualModelSetup + +## ParameterBounds +::: hapi.conceptual.ParameterBounds + +## Parameter-count helpers +::: hapi.conceptual.parameter_count + +::: hapi.conceptual.validate_parameter_count diff --git a/docs/api/period.md b/docs/api/period.md new file mode 100644 index 00000000..315d8fee --- /dev/null +++ b/docs/api/period.md @@ -0,0 +1,21 @@ +# Simulation period + +Six attributes on `Catchment` used to describe one thing: `start`, `end` and +`temporal_resolution` were given, and `date_index`, `dt` and `conversion_factor` were derived +from them in the constructor and then stored beside them as if they were independent. Storing a +derivation is how the three drift apart — reassigning `end` left `date_index` describing the old +span, with nothing to notice — and it is why the same `pd.date_range` branch was written out four +times across the package. + +`SimulationPeriod` holds the three inputs and derives the rest on read, so they cannot disagree. +It is frozen: a run covers the period it was built for, and a model that needs a different one +gets a new period rather than a mutated one. + +```python +model.period.date_index # one entry per step +model.period.days # how many steps +len(model.period) # the same number +``` + +## SimulationPeriod +::: hapi.period.SimulationPeriod diff --git a/docs/api/protocols.md b/docs/api/protocols.md new file mode 100644 index 00000000..5959f3cb --- /dev/null +++ b/docs/api/protocols.md @@ -0,0 +1,23 @@ +# Protocols + +`Run` does not import `Catchment`, and `Catchment` does not inherit from anything in the run +layer. What connects them is stated here: the run layer owns the interfaces, and `Catchment` +satisfies them structurally. + +That is dependency inversion, and it buys two checkable things. `hapi.run` and `hapi.wrapper` +carry no runtime dependency on the concrete class, so the arrow between the modules points the +other way. And the requirement is checked by mypy, where it used to live in prose in each +method's docstring — prose does not fail CI. + +`CatchmentLike` is deliberately builder-shaped: its fields really are optional, because a +half-built catchment is a legitimate state. Narrowing it into a run +(see [Runs](runs.md)) is where the optionality is resolved. + +## CatchmentLike +::: hapi.protocols.CatchmentLike + +## SupportsQsim +::: hapi.protocols.SupportsQsim + +## SpatialDistribution +::: hapi.protocols.SpatialDistribution diff --git a/docs/api/runs.md b/docs/api/runs.md new file mode 100644 index 00000000..f505e94d --- /dev/null +++ b/docs/api/runs.md @@ -0,0 +1,26 @@ +# Runs + +A `Catchment` is a **builder**: its inputs are `X | None` until the matching `read_*` call has +run, and that is honest. The engines need the opposite — a catchment that is finished. +`DistributedRun` and `LumpedRun` are that finished form. + +`from_model` is the single validation seam. Constructing a run *is* the validation: it resolves +every optional input, checks the drivers, the parameter cube, the river geometry and the +flow-path-length raster against the catchment grid, and refuses a combination the engines cannot +run. Every engine entry point takes one of these types, so the checks are enforced by the +signatures rather than by remembering to call them — which is how `Calibration`, going straight +to `Wrapper`, used to skip all of them. + +```python +run = DistributedRun.from_model(model) # checked here, once +results = Wrapper.run_muskingum(run) # nothing left to re-check +``` + +Both are frozen. The checks happen at construction, so a mutable run would let a caller swap an +input in afterwards and reach an engine with something never validated. + +## DistributedRun +::: hapi.runs.DistributedRun + +## LumpedRun +::: hapi.runs.LumpedRun diff --git a/mkdocs.yml b/mkdocs.yml index af93f7eb..bb03f39d 100644 --- a/mkdocs.yml +++ b/mkdocs.yml @@ -110,7 +110,11 @@ nav: - api/inputs.md - api/routing.md - api/run.md + - api/runs.md - api/results.md + - api/period.md + - api/conceptual.md + - api/protocols.md - api/wrapper.md - api/distrrm.md - api/hbv.md diff --git a/src/hapi/catchment.py b/src/hapi/catchment.py index 954e62be..be8bb248 100644 --- a/src/hapi/catchment.py +++ b/src/hapi/catchment.py @@ -51,15 +51,6 @@ if TYPE_CHECKING: from hapi.rrm.base_model import BaseConceptualModel -CONVERSION_FACTOR = (1000 * 24 * 60 * 60) / (1000**2) -#: (snow, maxbas) -> how many parameters the conceptual model reads in that configuration. -PARAMETER_COUNTS = { - (True, True): 16, - (False, True): 11, - (True, False): 17, - (False, False): 12, -} - #: Conceptual models `conceptual_model.model_class` can name in a YAML configuration. #: `read_lumped_model` still takes any `type[BaseConceptualModel]`, so this only bounds what the #: YAML shorthand can reach, not what the class accepts. From 8e8f9225dd4473c4a8bdb7a5d2b5d49e47eb6aa2 Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Thu, 10 Sep 2026 22:49:18 +0200 Subject: [PATCH 33/54] fix(inputs): close the rasters once they are read, and stop handing out result views Every raster read here opened a GDAL dataset and left it open. On Windows an open handle keeps a lock on the file, so a script that reads a catchment and then moves or rewrites its inputs fails, and a loop over basins -- or a repeated `results.save` -- accumulates handles for the life of the process. The four sites this branch wrote or rewrote now read inside a `with`: `SimulationResults._save_rasters`, `FlowNetwork.from_rasters` (two handles), `RiverGeometry.from_rasters` (five) and `Catchment.read_flow_path_length`. Everything they need is copied into arrays, so nothing wanted the handle afterwards. `_save_rasters` also handed `Datacube.values` a `np.moveaxis` view straight onto a result array. Nothing writes through it today, but a pyramids version that normalises no-data in place would edit the arrays a *save* is only supposed to read. It writes a contiguous copy. The related aliasing that stays is now stated where a reader will meet it: after a MAXBAS run `quz_routed` *is* `quz` rather than a copy of it, because the triangular routing works in place and a copy would double the memory of a `(rows, cols, time)` array for nothing. The field docs say so, since `results.quz_routed is results.quz` is otherwise invisible from outside. --- src/hapi/catchment.py | 17 ++++++++-------- src/hapi/inputs.py | 46 +++++++++++++++++++++++++------------------ src/hapi/results.py | 29 ++++++++++++++++++--------- 3 files changed, 56 insertions(+), 36 deletions(-) diff --git a/src/hapi/catchment.py b/src/hapi/catchment.py index be8bb248..24803d45 100644 --- a/src/hapi/catchment.py +++ b/src/hapi/catchment.py @@ -557,14 +557,15 @@ def read_flow_path_length(self, path: str): # Path validation is delegated to pyramids: a missing path raises # FileNotFoundError, a non-path argument TypeError, and an unreadable file a # GDAL RuntimeError. Unlike the asserts these replace, they survive `python -O`. - fpl = Dataset.read_file(path) - # No-data masking is delegated to pyramids (see FlowNetwork.from_rasters). The grid - # itself comes from the flow network, so this reader no longer redefines rows, cols, - # no_data_value or no_elem from a second raster. - self.flow_path_length_arr = np.ma.filled( - fpl.read_array(band=0, masked=True).astype(float), np.nan - ) - _warn_if_no_sentinel(fpl, "flow path length") + # Closed once read: see FlowNetwork.from_rasters for why the handle is not kept. + with Dataset.read_file(path) as fpl: + # No-data masking is delegated to pyramids (see FlowNetwork.from_rasters). The + # grid itself comes from the flow network, so this reader no longer redefines + # rows, cols, no_data_value or no_elem from a second raster. + self.flow_path_length_arr = np.ma.filled( + fpl.read_array(band=0, masked=True).astype(float), np.nan + ) + _warn_if_no_sentinel(fpl, "flow path length") logger.debug("Flow path length input is read successfully") diff --git a/src/hapi/inputs.py b/src/hapi/inputs.py index 59407399..0b858277 100644 --- a/src/hapi/inputs.py +++ b/src/hapi/inputs.py @@ -698,28 +698,31 @@ def from_rasters( UserWarning: A raster declares no no-data value, so every cell is treated as inside the catchment. """ - acc = Dataset.read_file(str(flow_acc)) - _warn_if_no_sentinel(acc, "flow accumulation") - acc_arr = np.ma.filled( - acc.read_array(band=0, masked=True).astype(float), np.nan - ) + # `with`: on Windows an open GDAL handle keeps a lock on the file, so a script + # that reads a catchment and then moves or rewrites its inputs fails, and a loop + # over basins accumulates handles. Everything read from the dataset is copied into + # arrays here, so nothing needs the handle afterwards. + with Dataset.read_file(str(flow_acc)) as acc: + _warn_if_no_sentinel(acc, "flow accumulation") + acc_arr = np.ma.filled( + acc.read_array(band=0, masked=True).astype(float), np.nan + ) + transform = acc.transform dir_arr, table = None, None if flow_dir is not None: - direction = DEM.read_file(str(flow_dir)) - _warn_if_no_sentinel(direction, "flow direction") - dir_arr = np.ma.filled( - direction.read_array(band=0, masked=True).astype(float), np.nan - ) - codes = set(np.unique(_to_int_codes(dir_arr)).tolist()) - if not codes <= set(D8_CODES): - raise ValueError( - "flow direction raster should contain values 1,2,4,8,16,32,64,128 " - f"only, found {sorted(codes - set(D8_CODES))}" + with DEM.read_file(str(flow_dir)) as direction: + _warn_if_no_sentinel(direction, "flow direction") + dir_arr = np.ma.filled( + direction.read_array(band=0, masked=True).astype(float), np.nan ) - table = direction.flow_direction_table() - - transform = acc.transform + codes = set(np.unique(_to_int_codes(dir_arr)).tolist()) + if not codes <= set(D8_CODES): + raise ValueError( + "flow direction raster should contain values 1,2,4,8,16,32,64,128 " + f"only, found {sorted(codes - set(D8_CODES))}" + ) + table = direction.flow_direction_table() network = cls( flow_acc_arr=acc_arr, flow_dir_arr=dir_arr, @@ -1797,7 +1800,12 @@ def from_rasters( river_roughness_file, floodplain_roughness_file, ) - return cls(*(Dataset.read_file(path).read_array(band=0) for path in paths)) + arrays = [] + for raster in paths: + # Five handles, closed as each raster is read: see FlowNetwork.from_rasters. + with Dataset.read_file(raster) as dataset: + arrays.append(dataset.read_array(band=0)) + return cls(*arrays) class Inputs: diff --git a/src/hapi/results.py b/src/hapi/results.py index 59aabbaf..524d979e 100644 --- a/src/hapi/results.py +++ b/src/hapi/results.py @@ -124,7 +124,13 @@ class SimulationResults: that will not look at it need not pay for it. See :attr:`~hapi.runs.DistributedRun.keep_state_variables`. quz_routed: Upper-zone discharge after routing. `None` until a routing step runs. - qlz_translated: Lower-zone discharge after translation. `None` until then. + After a MAXBAS run this *is* :attr:`quz`, not a copy of it -- the triangular + routing works in place and a copy would double the memory of a + `(rows, cols, time)` array for nothing. Nothing in the package writes through + the alias, but it is visible (`results.quz_routed is results.quz`), so editing + one in place edits the other. + qlz_translated: Lower-zone discharge after translation. `None` until then. Aliases + :attr:`qlz` after a MAXBAS run, for the same reason as :attr:`quz_routed`. q_total: `quz_routed + qlz_translated`. Read it through :attr:`outlet_shortcut_valid` rather than assuming what a cell means. qout: The outlet hydrograph, when the run computed one, and always `len(period)` @@ -677,8 +683,6 @@ def _save_rasters( arr = self._select(_RASTER_OPTIONS[result], start_i, end_i) - src = Dataset.read_file(flow_acc_path) - if prefix == "": prefix = "Result_" @@ -693,12 +697,19 @@ def _save_rasters( for i in period.date_index[start_i:end_i] ] - # from_dataset is pyramids' named constructor for an in-memory scaffold off a - # template raster; the bare Datacube(src, time_length=) form it replaced is kept - # only as a legacy fallback upstream. - cube = Datacube.from_dataset(src, arr.shape[2]) - cube.values = np.moveaxis(arr, -1, 0) - cube.to_file(names) + # Closed when the write finishes: on Windows an open GDAL handle keeps a lock on + # the file, so a script that saves rasters and then moves or deletes the template + # fails, and a repeated `save` accumulates handles. + with Dataset.read_file(flow_acc_path) as src: + # from_dataset is pyramids' named constructor for an in-memory scaffold off a + # template raster; the bare Datacube(src, time_length=) form it replaced is + # kept only as a legacy fallback upstream. + cube = Datacube.from_dataset(src, arr.shape[2]) + # A copy, not the `moveaxis` view: `arr` is a slice of a result array, and + # handing a view to a writer that may normalise no-data in place would edit + # the results this call is only supposed to read. + cube.values = np.ascontiguousarray(np.moveaxis(arr, -1, 0)) + cube.to_file(names) def _save_csv( self, From 036650c5dd2f3527fcccf9df97ec0c11e7be63b0 Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Thu, 10 Sep 2026 22:53:36 +0200 Subject: [PATCH 34/54] docs(examples): leave the pre-rename notebooks alone until they are migrated Eleven notebooks were swept by the CamelCase rename. All eleven still import the pre-rename `Hapi` package, so none of them can run and none of the edits changed anything that executes -- the `notebooks` pixi task's own description already records that they need migrating first. Four of them came out worse. `from Hapi.run import runHAPIwithLake` became `from Hapi.run import run_distributed_with_lake`, which turned an obviously stale line into one that looks current and still cannot work twice over: the package is `hapi`, and `run_distributed_with_lake` is a `staticmethod` on `Run`, never a module-level function. A reader can see that the first is old; the second reads as maintained. Reverted to their state on `main`. Migrating them properly -- package name, entry points, and the results object -- is its own piece of work. --- .../Jiboa-distributed-model-muskingum-lake.ipynb | 2 +- .../hydrological-model/Note books/Lumped-Model_Run.ipynb | 2 +- .../Note books/Lumped_Model_Calib.ipynb | 4 ++-- .../Note books/check-03Jiboa-colab.ipynb | 4 ++-- .../hydrological-model/Note books/check-03Jiboa.ipynb | 4 ++-- .../Note books/check-colab/Coello.ipynb | 2 +- .../Jiboa-distributed-model-muskingum-lake-colab.ipynb | 4 ++-- .../hydrological-model/Note books/check-colab/Jiboa.ipynb | 2 +- .../coello-distributed-model-run-muskingum.ipynb | 8 ++++---- .../Note books/coello-lumped-model-run-muskingum.ipynb | 2 +- .../Note books/lumped-model-run-coello.ipynb | 2 +- 11 files changed, 18 insertions(+), 18 deletions(-) diff --git a/examples/hydrological-model/Note books/Jiboa-distributed-model-muskingum-lake.ipynb b/examples/hydrological-model/Note books/Jiboa-distributed-model-muskingum-lake.ipynb index 2461a6f1..2d00c7ad 100644 --- a/examples/hydrological-model/Note books/Jiboa-distributed-model-muskingum-lake.ipynb +++ b/examples/hydrological-model/Note books/Jiboa-distributed-model-muskingum-lake.ipynb @@ -242,7 +242,7 @@ "metadata": {}, "outputs": [], "source": [ - "Run.run_distributed_with_lake(Jiboa, JiboaLake)" + "Run.runHAPIwithLake(Jiboa, JiboaLake)" ] }, { diff --git a/examples/hydrological-model/Note books/Lumped-Model_Run.ipynb b/examples/hydrological-model/Note books/Lumped-Model_Run.ipynb index c6a624f7..87ba5bc3 100644 --- a/examples/hydrological-model/Note books/Lumped-Model_Run.ipynb +++ b/examples/hydrological-model/Note books/Lumped-Model_Run.ipynb @@ -208,7 +208,7 @@ "metadata": {}, "outputs": [], "source": [ - "Run.run_lumped(Coello, Route, RoutingFn)" + "Run.runLumped(Coello, Route, RoutingFn)" ] }, { diff --git a/examples/hydrological-model/Note books/Lumped_Model_Calib.ipynb b/examples/hydrological-model/Note books/Lumped_Model_Calib.ipynb index b7c7b924..9aa02740 100644 --- a/examples/hydrological-model/Note books/Lumped_Model_Calib.ipynb +++ b/examples/hydrological-model/Note books/Lumped_Model_Calib.ipynb @@ -261,7 +261,7 @@ "metadata": {}, "outputs": [], "source": [ - "cal_parameters = Coello.calibrate_lumped(\n", + "cal_parameters = Coello.lumpedCalibration(\n", " Basic_inputs, OptimizationArgs, printError=None\n", ")\n", "\n", @@ -296,7 +296,7 @@ "outputs": [], "source": [ "Coello.parameters = cal_parameters[1]\n", - "Run.run_lumped(Coello, Route, RoutingFn)" + "Run.runLumped(Coello, Route, RoutingFn)" ] }, { diff --git a/examples/hydrological-model/Note books/check-03Jiboa-colab.ipynb b/examples/hydrological-model/Note books/check-03Jiboa-colab.ipynb index 7154062d..241121db 100644 --- a/examples/hydrological-model/Note books/check-03Jiboa-colab.ipynb +++ b/examples/hydrological-model/Note books/check-03Jiboa-colab.ipynb @@ -127,7 +127,7 @@ "import pandas as pd\n", "\n", "# HAPI modules\n", - "from Hapi.run import run_distributed_with_lake" + "from Hapi.run import runHAPIwithLake" ] }, { @@ -258,7 +258,7 @@ "outputs": [], "source": [ "Sim = pd.DataFrame(index=lakeCalib.index)\n", - "st, Sim['Q'], q_uz_routed, q_lz_trans = run_distributed_with_lake(\n", + "st, Sim['Q'], q_uz_routed, q_lz_trans = runHAPIwithLake(\n", " HBV,\n", " Paths,\n", " ParPath,\n", diff --git a/examples/hydrological-model/Note books/check-03Jiboa.ipynb b/examples/hydrological-model/Note books/check-03Jiboa.ipynb index 0c77a680..b922dc24 100644 --- a/examples/hydrological-model/Note books/check-03Jiboa.ipynb +++ b/examples/hydrological-model/Note books/check-03Jiboa.ipynb @@ -55,7 +55,7 @@ "import pandas as pd\n", "\n", "# HAPI modules\n", - "from Hapi.rrm import run_distributed_with_lake\n", + "from Hapi.rrm import runHAPIwithLake\n", "from osgeo import gdal" ] }, @@ -137,7 +137,7 @@ "outputs": [], "source": [ "Sim = pd.DataFrame(index=lakeCalib.index)\n", - "st, Sim['Q'], q_uz_routed, q_lz_trans = run_distributed_with_lake(\n", + "st, Sim['Q'], q_uz_routed, q_lz_trans = runHAPIwithLake(\n", " HBV,\n", " Paths,\n", " ParPath,\n", diff --git a/examples/hydrological-model/Note books/check-colab/Coello.ipynb b/examples/hydrological-model/Note books/check-colab/Coello.ipynb index f516152d..85f2a341 100644 --- a/examples/hydrological-model/Note books/check-colab/Coello.ipynb +++ b/examples/hydrological-model/Note books/check-colab/Coello.ipynb @@ -233,7 +233,7 @@ }, "outputs": [], "source": [ - "Run.run_distributed(Coello)" + "Run.RunHapi(Coello)" ] }, { diff --git a/examples/hydrological-model/Note books/check-colab/Jiboa-distributed-model-muskingum-lake-colab.ipynb b/examples/hydrological-model/Note books/check-colab/Jiboa-distributed-model-muskingum-lake-colab.ipynb index c04ecd27..6852aeeb 100644 --- a/examples/hydrological-model/Note books/check-colab/Jiboa-distributed-model-muskingum-lake-colab.ipynb +++ b/examples/hydrological-model/Note books/check-colab/Jiboa-distributed-model-muskingum-lake-colab.ipynb @@ -126,7 +126,7 @@ "import pandas as pd\n", "\n", "# HAPI modules\n", - "from Hapi.run import run_distributed_with_lake\n", + "from Hapi.run import runHAPIwithLake\n", "from osgeo import gdal" ] }, @@ -258,7 +258,7 @@ "outputs": [], "source": [ "Sim = pd.DataFrame(index=lakeCalib.index)\n", - "st, Sim['Q'], q_uz_routed, q_lz_trans = run_distributed_with_lake(\n", + "st, Sim['Q'], q_uz_routed, q_lz_trans = runHAPIwithLake(\n", " HBV,\n", " Paths,\n", " ParPath,\n", diff --git a/examples/hydrological-model/Note books/check-colab/Jiboa.ipynb b/examples/hydrological-model/Note books/check-colab/Jiboa.ipynb index 509b2fa7..4606a678 100644 --- a/examples/hydrological-model/Note books/check-colab/Jiboa.ipynb +++ b/examples/hydrological-model/Note books/check-colab/Jiboa.ipynb @@ -287,7 +287,7 @@ "outputs": [], "source": [ "Sim = pd.DataFrame(index=lakeCalib.index)\n", - "st, Sim['Q'], q_uz_routed, q_lz_trans = run_distributed_with_lake(\n", + "st, Sim['Q'], q_uz_routed, q_lz_trans = runHAPIwithLake(\n", " HBV,\n", " Paths,\n", " ParPath,\n", diff --git a/examples/hydrological-model/Note books/coello-distributed-model-run-muskingum.ipynb b/examples/hydrological-model/Note books/coello-distributed-model-run-muskingum.ipynb index 00d72f28..dc5a61ec 100644 --- a/examples/hydrological-model/Note books/coello-distributed-model-run-muskingum.ipynb +++ b/examples/hydrological-model/Note books/coello-distributed-model-run-muskingum.ipynb @@ -184,7 +184,7 @@ "metadata": {}, "outputs": [], "source": [ - "Run.run_distributed(Coello)" + "Run.RunHapi(Coello)" ] }, { @@ -196,7 +196,7 @@ "source": [ "import numpy as np\n", "\n", - "np.shape(Coello.results.q_total)" + "np.shape(Coello.Qtot)" ] }, { @@ -248,7 +248,7 @@ "metadata": {}, "outputs": [], "source": [ - "Coello.results.q_total[0, 0, 0]" + "Coello.Qtot[0, 0, 0]" ] }, { @@ -300,7 +300,7 @@ "source": [ "import matplotlib.pyplot as plt\n", "\n", - "plt.plot(Coello.results.q_total[12, 1, :])" + "plt.plot(Coello.Qtot[12, 1, :])" ] }, { diff --git a/examples/hydrological-model/Note books/coello-lumped-model-run-muskingum.ipynb b/examples/hydrological-model/Note books/coello-lumped-model-run-muskingum.ipynb index be7681e1..03232e66 100644 --- a/examples/hydrological-model/Note books/coello-lumped-model-run-muskingum.ipynb +++ b/examples/hydrological-model/Note books/coello-lumped-model-run-muskingum.ipynb @@ -243,7 +243,7 @@ "metadata": {}, "outputs": [], "source": [ - "Run.run_lumped(Coello, Route, RoutingFn)" + "Run.runLumped(Coello, Route, RoutingFn)" ] }, { diff --git a/examples/hydrological-model/Note books/lumped-model-run-coello.ipynb b/examples/hydrological-model/Note books/lumped-model-run-coello.ipynb index bc64070b..8d30f9c8 100644 --- a/examples/hydrological-model/Note books/lumped-model-run-coello.ipynb +++ b/examples/hydrological-model/Note books/lumped-model-run-coello.ipynb @@ -229,7 +229,7 @@ "metadata": {}, "outputs": [], "source": [ - "Run.run_lumped(Coello, Route, RoutingFn)" + "Run.runLumped(Coello, Route, RoutingFn)" ] }, { From 5136103cca2ed4f215c044b3d4a6048895023b9e Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Thu, 10 Sep 2026 22:53:36 +0200 Subject: [PATCH 35/54] test: rename the classes and tests that still name the removed API The branch's stated position is that everything is `snake_case` everywhere and that `_maxbas_routed` is gone. The test suite still carried `TestDistMaxbas2`, `TestFW1Calibration`, `TestRunFloodModel`, `TestRunHapiWithLake`, `TestRunHapiWithLakeEndToEnd`, `test_run_fw1_returns_maxbas_routed_results` and a `test_marks_the_model_as_maxbas_routed` whose docstring still explained what the flag did -- which is the first place someone greps for a name they cannot find. Renamed to the entry points that survive, and the two docstrings that described the flag now describe the routing recorded on the results. The remaining mentions of `_maxbas_routed` are in module docstrings explaining what it was replaced by, which is the one place the old name still earns its keep. --- .../calibration/test_calibration_distributed.py | 2 +- .../rrm/catchment/test_maxbas_routing_variants.py | 2 +- tests/rrm/catchment/test_run_results_coupling.py | 2 +- tests/rrm/catchment/test_run_validation.py | 4 ++-- tests/rrm/catchment/test_wrapper_with_lake.py | 14 ++++++++------ 5 files changed, 13 insertions(+), 11 deletions(-) diff --git a/tests/rrm/calibration/test_calibration_distributed.py b/tests/rrm/calibration/test_calibration_distributed.py index a2df16f2..650fb65a 100644 --- a/tests/rrm/calibration/test_calibration_distributed.py +++ b/tests/rrm/calibration/test_calibration_distributed.py @@ -397,7 +397,7 @@ def spy(model, *args, **kwargs): ) -class TestFW1Calibration: +class TestCalibrateMaxbas: """Tests for `Calibration.calibrate_maxbas` (triangular routing).""" def test_stores_the_optimizer_result_on_the_instance( diff --git a/tests/rrm/catchment/test_maxbas_routing_variants.py b/tests/rrm/catchment/test_maxbas_routing_variants.py index 979a4413..1af968b4 100644 --- a/tests/rrm/catchment/test_maxbas_routing_variants.py +++ b/tests/rrm/catchment/test_maxbas_routing_variants.py @@ -91,7 +91,7 @@ def maxbas_parameters_path(lumped_parameters_path: str, tmp_path_factory) -> str return str(path) -class TestDistMaxbas2: +class TestRouteMaxbasByPathLength: """Tests for `DistributedRRM.route_maxbas_by_path_length`.""" def test_conserves_volume_while_redistributing_it_in_time( diff --git a/tests/rrm/catchment/test_run_results_coupling.py b/tests/rrm/catchment/test_run_results_coupling.py index a5af90a0..9b63350f 100644 --- a/tests/rrm/catchment/test_run_results_coupling.py +++ b/tests/rrm/catchment/test_run_results_coupling.py @@ -251,7 +251,7 @@ def test_run_hapi_returns_the_object_it_put_on_the_model( f"a run_distributed run is Muskingum-routed, got {results.routing}" ) - def test_run_fw1_returns_maxbas_routed_results( + def test_run_maxbas_returns_maxbas_routed_results( self, coello_fixtures: dict, coello_dist_parameters_maxbas: str ): """Test that the triangular path records MAXBAS on the results it returns. diff --git a/tests/rrm/catchment/test_run_validation.py b/tests/rrm/catchment/test_run_validation.py index dff174ac..0caf1c14 100644 --- a/tests/rrm/catchment/test_run_validation.py +++ b/tests/rrm/catchment/test_run_validation.py @@ -108,7 +108,7 @@ def _load_flat_river_geometry(model: Catchment) -> None: ) -class TestRunFloodModel: +class TestRunFlood: """Tests for `Run.run_flood`.""" def test_dispatches_once_every_input_lines_up( @@ -184,7 +184,7 @@ def test_rejects_meteo_that_does_not_cover_the_grid( ) -class TestRunHapiWithLake: +class TestRunDistributedWithLake: """Tests for `Run.run_distributed_with_lake`.""" def test_dispatches_once_the_lake_record_matches_the_simulation( diff --git a/tests/rrm/catchment/test_wrapper_with_lake.py b/tests/rrm/catchment/test_wrapper_with_lake.py index b57b58a0..4a55a79d 100644 --- a/tests/rrm/catchment/test_wrapper_with_lake.py +++ b/tests/rrm/catchment/test_wrapper_with_lake.py @@ -3,7 +3,7 @@ `Wrapper.run_muskingum_with_lake` and `Wrapper.run_maxbas_with_lake` run the lake as a lumped inflow, add its routed discharge to the outflow cell, and then route the sub-catchment — Muskingum in the first case, triangular (MAXBAS) in the second. Neither had test coverage, so the sizes they read off -`MeteoInputs` and `FlowNetwork` and the `_maxbas_routed` flag they leave behind were unpinned. +`MeteoInputs` and `FlowNetwork` and the routing they record on their results were unpinned. The bundled Jiboa lake fixture is hourly and twenty steps long while its distributed rasters are absent from the repository, so the lake here is driven over the Coello grid instead: real @@ -479,18 +479,20 @@ def test_the_outlet_series_carries_the_lake_and_drops_the_extra_slot( err_msg="qout must be the trimmed sub-catchment sum plus the trimmed lake series", ) - def test_marks_the_model_as_maxbas_routed( + def test_records_maxbas_on_the_results( self, coello_with_lake_inputs_maxbas: Catchment, coello_start_date: str, coello_end_date: str, ): - """Test that the triangular lake path sets `_maxbas_routed` from a clean model. + """Test that the triangular lake path records MAXBAS on the results it returns. Test scenario: Triangular routing sends every cell straight to the outlet, so reading a gauge - cell of `q_total` under-reports. The flag is what makes `extract_discharge` refuse - rather than return the wrong hydrograph. + cell of `q_total` under-reports. `RoutingKind.MAXBAS` on the arrays is what makes + `extract_discharge` refuse rather than return the wrong hydrograph -- it replaced + a `_maxbas_routed` flag on the catchment that three methods had to set and clear + by hand. """ model = coello_with_lake_inputs_maxbas lake = _make_lake(model, coello_start_date, coello_end_date, seed=23) @@ -505,7 +507,7 @@ def test_marks_the_model_as_maxbas_routed( ) -class TestRunHapiWithLakeEndToEnd: +class TestRunDistributedWithLakeEndToEnd: """Tests that the public entry point completes with the record it documents.""" def test_the_entry_point_runs_with_a_record_of_the_documented_length( From fc62305275f8595e202d1a9a9f5ca2069b5ebde4 Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Thu, 10 Sep 2026 22:53:51 +0200 Subject: [PATCH 36/54] fix(results): correct the animate option list, and the dtype the mask assumes Six small things the review turned up, all in code or prose this branch wrote. `animate`'s docstring listed option 2 as "Upper zone discharge" and option 3 as "Ground water" while the options themselves are titled "Surface Flow" and "Ground Water Flow" and read `quz_routed` / `qlz_translated`. The list now says what the options do, and which ranges are states and drivers. The no-data mask writes NaN into a copy of the selected array. The meteo options read `MeteoInputs` cubes, which are documented as carried through "as stored", so an integer driver raster would raise `cannot convert float NaN to integer`. It copies to float32 when the dtype cannot hold NaN, and leaves float arrays alone rather than upcasting them. `_save_csv` tested the same kind of membership two different ways in adjacent lines. `read_lumped_model` instantiates the conceptual model, so `ConceptualModelSetup` holds an *instance*; the `save` doctest passed the class, which happened to work because nothing called it, and contradicted the example in `conceptual.py`. `docs/api/catchment.md` said "up to and including version 1.7.0" about a check this branch adds, while `pyproject.toml` still reads 1.7.0 -- it described the current release as the past one. Three added lines were over the repository's 120-character limit. --- docs/api/catchment.md | 7 +++--- docs/examples/distributed-model-calib.md | 3 ++- docs/examples/lumped-model-run.md | 3 ++- src/hapi/results.py | 25 +++++++++++++-------- tests/calibration/distributed_mode_calib.py | 5 ++++- 5 files changed, 27 insertions(+), 16 deletions(-) diff --git a/docs/api/catchment.md b/docs/api/catchment.md index 8a664bd7..fbe35a2d 100644 --- a/docs/api/catchment.md +++ b/docs/api/catchment.md @@ -11,10 +11,9 @@ and stored in the one spelling the internals compare against: | `maxbas` | `MAXBAS` | Every cell straight to the outlet through a triangular function. | | `kinematic` | `Kinematic` | The flood model's own path (`Run.run_flood`). | -Anything else raises a `ValueError` naming the three. Up to and including version 1.7.0 the -constructor stored whatever string it was handed, so a run configured as `"Max_bas"` — or as a -descriptive label such -as `"Muskingum-Cunge"` — was accepted and then silently routed with Muskingum, because +Anything else raises a `ValueError` naming the three. Before this check the constructor stored +whatever string it was handed, so a run configured as `"Max_bas"` — or as a descriptive label +such as `"Muskingum-Cunge"` — was accepted and then silently routed with Muskingum, because `distrrm.route_muskingum` compares against `"Muskingum"` exactly. Rejecting the spelling is what makes that comparison trustworthy; a script passing a spelling outside the table has to be updated to one of the three. diff --git a/docs/examples/distributed-model-calib.md b/docs/examples/distributed-model-calib.md index 7cc31c9a..50630b9f 100644 --- a/docs/examples/distributed-model-calib.md +++ b/docs/examples/distributed-model-calib.md @@ -100,7 +100,8 @@ Coello.read_discharge_gauges(GaugesPath, column='id', fmt="%Y-%m-%d") - The `Run` object connects all the components of the simulation together, the `Catchment` object, the `Lake` object and the `distributedrouting` object -- import the Run object and use the `Catchment` object as a parameter to the `Run` object, then call the run_distributed method to start the simulation +- import the Run object and use the `Catchment` object as a parameter to the `Run` + object, then call the run_distributed method to start the simulation ```python from hapi.run import Run diff --git a/docs/examples/lumped-model-run.md b/docs/examples/lumped-model-run.md index 40327329..9092803e 100644 --- a/docs/examples/lumped-model-run.md +++ b/docs/examples/lumped-model-run.md @@ -64,7 +64,8 @@ Coello.read_parameters(Parameterpath, Snow) RoutingFn = Routing.muskingum_v Route = 1 ``` -- now all the data required for the model are prepared in the right form, now you can call the `run_lumped` wrapper to initiate the calculation +- now all the data required for the model are prepared in the right form, now you can + call the `run_lumped` wrapper to initiate the calculation ```python Run.run_lumped(Coello, Route, RoutingFn) diff --git a/src/hapi/results.py b/src/hapi/results.py index 524d979e..d0618eba 100644 --- a/src/hapi/results.py +++ b/src/hapi/results.py @@ -387,10 +387,11 @@ def animate( start: Starting date of the animation. end: End date of the animation. fmt: Format a string date is read with. Default is "%Y-%m-%d". - option: Variable to animate. 1 - Total discharge, 2 - Upper zone discharge, - 3 - Ground water, 4 - Snow pack, 5 - Soil moisture, 6 - Upper zone, - 7 - Lower zone, 8 - Water content, 9 - Precipitation, 10 - ET, - 11 - Temperature. Default is 1. + option: Variable to animate. 1 - Total discharge, 2 - Surface flow (the routed + upper zone), 3 - Ground water flow (the translated lower zone), 4 - Snow + pack, 5 - Soil moisture, 6 - Upper zone, 7 - Lower zone, 8 - Water content, + 9 - Precipitation, 10 - ET, 11 - Temperature. Default is 1. Options 4-8 are + the state variables and 9-11 are the run's own drivers. gauges: Gauge table to overlay, as `Catchment.GaugesTable`. It must carry `id`, `cell_row` and `cell_col` columns. `None`, the default, draws no gauges. This used to be a `bool` that reached back onto the catchment for the table; @@ -462,8 +463,14 @@ def animate( source, title = _ANIMATION_OPTIONS[option] arr = self._select(source, start_i, end_i) - # mask the no-data cells on a copy so plotting never mutates the result arrays - arr = arr.copy() + # Masked on a copy, so plotting never mutates the result arrays -- and on a float + # copy, because the mask writes NaN and the meteo options read `MeteoInputs` cubes + # "as stored", which an integer driver raster would make unassignable. + arr = ( + arr.copy() + if np.issubdtype(arr.dtype, np.floating) + else arr.astype(np.float32) + ) arr[np.isnan(run.flow_network.flow_acc_arr), :] = np.nan time = run.period.date_index[start_i:end_i] @@ -592,7 +599,7 @@ def save( ... data=np.ones((len(period), 4)), ... parameters=ParameterSet(np.ones(12), snow=False, maxbas=False), ... model_setup=ConceptualModelSetup( - ... HBVBergestrom92, 100.0, [0.0] * 5, 1.0 + ... HBVBergestrom92(), 100.0, [0.0] * 5, 1.0 ... ), ... ) >>> discharge = np.array([1.5, 2.5, 3.5]) @@ -746,9 +753,9 @@ def _save_csv( # For a lumped run the total discharge *is* `Qsim`; `Run.run_lumped` only wraps # this same array in a frame to put on the model. data["Qsim"] = self._require_field("q_total")[start_i:end_i] - if result == 2 or result == 5: + if result in (2, 5): data["Quz"] = self.quz[start_i:end_i] - if result == 3 or result == 5: + if result in (3, 5): data["Qlz"] = self.qlz[start_i:end_i] if result in (4, 5): data[STATE_VARIABLES] = self._require_state_variables()[start_i:end_i, :] diff --git a/tests/calibration/distributed_mode_calib.py b/tests/calibration/distributed_mode_calib.py index cdb2c058..03a02335 100644 --- a/tests/calibration/distributed_mode_calib.py +++ b/tests/calibration/distributed_mode_calib.py @@ -139,7 +139,10 @@ def objective_function(Qobs, Qout, q_uz_routed, q_lz_trans, coordinates): spatial_var_fun, optimization_args, print_error=0 ) # %% convert parameters to rasters -# Coello.model.parameters.values = [0.700, 399, 1.704, 0.1021, 0.4622, 0.6237, 0.1251, 0.005, 59.85, 5.241, 94.91, 0.2075] +# Coello.model.parameters = Coello.model.parameters.with_values( +# [0.700, 399, 1.704, 0.1021, 0.4622, 0.6237, 0.1251, 0.005, 59.85, 5.241, +# 94.91, 0.2075] +# ) spatial_var_fun.Function( Coello.model.parameters.values, kub=spatial_var_fun.Kub, klb=spatial_var_fun.Klb ) From 1df51e52b1dddf184ba7b346f4f6217038f2ac76 Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Thu, 10 Sep 2026 22:54:37 +0200 Subject: [PATCH 37/54] style(tests): sort the inspect import into the stdlib block --- tests/rrm/catchment/test_config.py | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/tests/rrm/catchment/test_config.py b/tests/rrm/catchment/test_config.py index c30ed170..380abe9f 100644 --- a/tests/rrm/catchment/test_config.py +++ b/tests/rrm/catchment/test_config.py @@ -16,11 +16,10 @@ import copy import datetime as dt +import inspect import os from pathlib import Path -import inspect - import pytest import yaml from pydantic import ValidationError From 67e5de042409eeff0f459384dd33e2aad836e78c Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Thu, 10 Sep 2026 22:55:05 +0200 Subject: [PATCH 38/54] style: apply ruff-format to the migrated calibration scripts and tests --- ...el-calibration-deap-multiobjective-NSE-NSEHF.py | 4 ++-- ...del-calibration-deap-multiobjective-NSE-RMSE.py | 4 ++-- .../coello-lumped-model-calibration-deap.py | 14 ++++++++++---- tests/rrm/calibration/test_rrm_calibration.py | 4 +--- 4 files changed, 15 insertions(+), 11 deletions(-) diff --git a/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-NSEHF.py b/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-NSEHF.py index 976deba8..4edea0e8 100644 --- a/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-NSEHF.py +++ b/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-NSEHF.py @@ -97,8 +97,8 @@ def initializer(): def objfn(individual): # Coello.model.read_parameters(Parameterpath, Snow) Coello.model.parameters = ParameterSet( - individual, snow=Coello.bounds.snow, maxbas=Coello.bounds.maxbas -) + individual, snow=Coello.bounds.snow, maxbas=Coello.bounds.maxbas + ) Run.run_lumped(Coello.model, Route, RoutingFn) # [Coello.model.QGauges.columns[-1]] NSE = metrics.nse_hf(Coello.model.QGauges, Coello.model.Qsim, *Coello.OFArgs) diff --git a/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-RMSE.py b/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-RMSE.py index eaa7f494..d71f8b79 100644 --- a/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-RMSE.py +++ b/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap-multiobjective-NSE-RMSE.py @@ -99,8 +99,8 @@ def initializer(): def objfn(individual): # Coello.model.read_parameters(Parameterpath, Snow) Coello.model.parameters = ParameterSet( - individual, snow=Coello.bounds.snow, maxbas=Coello.bounds.maxbas -) + individual, snow=Coello.bounds.snow, maxbas=Coello.bounds.maxbas + ) Run.run_lumped(Coello.model, Route, RoutingFn) # [Coello.model.QGauges.columns[-1]] NSE = metrics.nse_hf(Coello.model.QGauges, Coello.model.Qsim, *Coello.OFArgs) diff --git a/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap.py b/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap.py index 05b60b6c..b6346697 100644 --- a/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap.py +++ b/examples/hydrological-model/coello/calibration/coello-lumped-model-calibration-deap.py @@ -95,8 +95,8 @@ def initializer(): def objfn(individual): # Coello.model.read_parameters(Parameterpath, Snow) Coello.model.parameters = ParameterSet( - individual, snow=Coello.bounds.snow, maxbas=Coello.bounds.maxbas -) + individual, snow=Coello.bounds.snow, maxbas=Coello.bounds.maxbas + ) Run.run_lumped(Coello.model, Route, RoutingFn) # [Coello.model.QGauges.columns[-1]] error = PC.NSEHF(Coello.model.QGauges, Coello.model.Qsim, *Coello.OFArgs) @@ -184,7 +184,10 @@ def distance(individual): # %% Save the Parameters ParPath = ( - Path + f"{Coello.model.name}-lumped-parameters" + str(dt.datetime.now())[0:10] + ".txt" + Path + + f"{Coello.model.name}-lumped-parameters" + + str(dt.datetime.now())[0:10] + + ".txt" ) parameters = pd.DataFrame(index=parnames) # parameters['values'] = cal_parameters[1] @@ -196,6 +199,9 @@ def distance(individual): EndDate = "2010-04-20" Path = ( - Path + f"{Coello.model.name}-results-lumped-model" + str(dt.datetime.now())[0:10] + ".txt" + Path + + f"{Coello.model.name}-results-lumped-model" + + str(dt.datetime.now())[0:10] + + ".txt" ) Coello.model.results.save(result=5, start=StartDate, end=EndDate, path=Path) diff --git a/tests/rrm/calibration/test_rrm_calibration.py b/tests/rrm/calibration/test_rrm_calibration.py index cdfb1743..928291d0 100644 --- a/tests/rrm/calibration/test_rrm_calibration.py +++ b/tests/rrm/calibration/test_rrm_calibration.py @@ -50,9 +50,7 @@ def test_read_parameters_bounds_refuses_a_width_the_configuration_does_not_call_ coello = Calibration(Catchment("rrm", coello_rrm_date[0], coello_rrm_date[1])) with pytest.raises(ValueError): - coello.read_parameters_bound( - [0.0] * width, [1.0] * width, snow, maxbas=maxbas - ) + coello.read_parameters_bound([0.0] * width, [1.0] * width, snow, maxbas=maxbas) def test_lumped_calibration( From cdecd644a304ecadccaa57c269732bab985186c4 Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Thu, 10 Sep 2026 23:10:45 +0200 Subject: [PATCH 39/54] test: cover every guard this round's review fixes touched Round 1's fixes added or changed twelve guards, and coverage on the modules they live in ranged from 91% to 96% -- the new checks were exercised, but the branches around them were not. All seven are at 100% line and branch now: conceptual 93% -> 100% runs 92% -> 100% period 96% -> 100% wrapper 91% -> 100% distrrm 96% -> 100% run 96% -> 100% results 100% -> 100% What was untested and now is: the parameter cube's *columns* check (only rows had a test, so half the guard was unexercised); the skip-without-geometry guard on `DistributedRun.__post_init__`, which `from_model` had always refused earlier; a direction raster carrying no lookup table; `routing_table` on a network built without one; a backwards simulation span; an initial condition that is not five values; bounds of different lengths; `ParameterSet.count`; the path-length router with no raster; lumped routing asked for with something that is not callable; a lake with no record at all; and each of the three lake inputs `_lake_inputs` names. One of these is worth more than the coverage number: nothing exercised the *taken* branch of `skip_hydraulic_cells`. The flood tests check that the warning fires, not that the cells come out unrouted -- and the cell has to be downstream, because a headwater is copied across before the skip is consulted. That test now pins the handoff itself. `if __name__ == "__main__": print("Wrapper")` and its twin in `run.py` are deleted rather than excluded: vestigial module-run scaffolding that printed a class name, the same thing the do-nothing `__init__`s were. 685 tests in the coverage run, 668 in the main task plus 17 in `plot`; doctests, mypy, ruff all clean. --- src/hapi/run.py | 4 - src/hapi/wrapper.py | 4 - .../catchment/test_maxbas_routing_variants.py | 87 ++++++++++++++++++ .../test_read_parameters_validation.py | 54 +++++++++++ tests/rrm/catchment/test_results.py | 48 +++++++++- tests/rrm/catchment/test_run_narrowing.py | 90 +++++++++++++++++++ tests/rrm/catchment/test_run_validation.py | 22 +++++ tests/rrm/catchment/test_wrapper_with_lake.py | 42 +++++++++ 8 files changed, 342 insertions(+), 9 deletions(-) diff --git a/src/hapi/run.py b/src/hapi/run.py index be10f8a4..f526cd39 100644 --- a/src/hapi/run.py +++ b/src/hapi/run.py @@ -384,7 +384,3 @@ def run_lumped( model.results = results logger.info("Lumped model run has finished successfully") return results - - -if __name__ == "__main__": - print("Run") diff --git a/src/hapi/wrapper.py b/src/hapi/wrapper.py index 4a4280ae..e802eb0c 100644 --- a/src/hapi/wrapper.py +++ b/src/hapi/wrapper.py @@ -408,7 +408,3 @@ def run_lumped( ) results.q_total = q_total return results - - -if __name__ == "__main__": - print("Wrapper") diff --git a/tests/rrm/catchment/test_maxbas_routing_variants.py b/tests/rrm/catchment/test_maxbas_routing_variants.py index 1af968b4..a2ba49ea 100644 --- a/tests/rrm/catchment/test_maxbas_routing_variants.py +++ b/tests/rrm/catchment/test_maxbas_routing_variants.py @@ -213,6 +213,93 @@ def _lumped_model( return model +class TestRoutingGuards: + """The two routing entry points that refuse rather than run on missing inputs.""" + + def test_path_length_routing_says_when_it_has_no_raster( + self, + coello_start_date: str, + coello_end_date: str, + coello_prec_path: str, + coello_temp_path: str, + coello_evap_path: str, + coello_acc_path: str, + coello_dist_parameters_maxbas: str, + coello_cat_area: int, + coello_initial_cond: list, + ): + """Test that scaling MAXBAS by path length without the raster names the reader. + + Test scenario: + `route_maxbas_by_path_length` is the only entry point that reads + `flow_path_length`, and a run is perfectly valid without one. Without the guard + this was a `TypeError` on `None` inside `np.nanmax`. + """ + model = Catchment( + "coello", + coello_start_date, + coello_end_date, + spatial_resolution="Distributed", + temporal_resolution="Daily", + ) + model.meteo = MeteoInputs.from_rasters( + coello_prec_path, + coello_temp_path, + coello_evap_path, + start=coello_start_date, + end=coello_end_date, + regex_string=r"\d{4}.\d{2}.\d{2}", + date=True, + file_name_data_fmt="%Y.%m.%d", + ) + model.flow_network = FlowNetwork.from_rasters(coello_acc_path) + model.read_parameters(coello_dist_parameters_maxbas, False, maxbas=True) + model.read_lumped_model(HBVLumped, coello_cat_area, coello_initial_cond) + run = DistributedRun.from_model(model, needs_flow_direction=False) + results = DistributedRRM.run_lumped_model(run) + + with pytest.raises(ValueError, match="flow-path-length raster"): + DistributedRRM.route_maxbas_by_path_length(run, results) + + @pytest.mark.parametrize("routing_fn", [None, "not a function", 7]) + def test_lumped_routing_refuses_something_that_cannot_be_called( + self, + coello_rrm_date: list, + lumped_meteo_data_path: str, + maxbas_parameters_path: str, + coello_AreaCoeff: float, + coello_InitialCond: list, + routing_fn, + ): + """Test that asking for routing without a callable is refused by name. + + Args: + coello_rrm_date: Start and end dates. + lumped_meteo_data_path: Driver record. + maxbas_parameters_path: Parameter file. + coello_AreaCoeff: Catchment area. + coello_InitialCond: Initial state. + routing_fn: Something that is not a routing function. + + Test scenario: + `Route` and `routing_fn` are separate arguments, so they can disagree -- and a + docs page had already passed the function as the flag. Without the guard the + failure is a `TypeError: 'NoneType' object is not callable` deep in the wrapper. + """ + model = _lumped_model( + coello_rrm_date, + lumped_meteo_data_path, + maxbas_parameters_path, + coello_AreaCoeff, + coello_InitialCond, + ) + + with pytest.raises(TypeError, match="callable"): + Wrapper.run_lumped( + LumpedRun.from_model(model), Routing=1, RoutingFn=routing_fn + ) + + class TestLumpedRouting: """Tests for the routing branches of `Wrapper.run_lumped` reached through `Run.run_lumped`.""" diff --git a/tests/rrm/catchment/test_read_parameters_validation.py b/tests/rrm/catchment/test_read_parameters_validation.py index 0f4b5046..24423571 100644 --- a/tests/rrm/catchment/test_read_parameters_validation.py +++ b/tests/rrm/catchment/test_read_parameters_validation.py @@ -13,6 +13,7 @@ import pytest from hapi.catchment import Catchment +from hapi.conceptual import ParameterBounds, ParameterSet from hapi.period import SimulationPeriod from hapi.rrm.hbv_bergestrom92 import HBVBergestrom92 as HBVLumped @@ -86,6 +87,59 @@ def test_unknown_resolution_is_rejected_at_construction(self): with pytest.raises(ValueError, match="temporal resolutions"): Catchment("coello", "2009-01-01", "2009-01-10", temporal_resolution="15min") + @pytest.mark.parametrize("initial_cond", [[0, 10, 10], [0] * 7, []]) + def test_an_initial_condition_of_the_wrong_length_is_refused( + self, initial_cond: list + ): + """Test that the initial state must carry exactly the five state variables. + + Args: + initial_cond: A state list of the wrong length. + + Test scenario: + The five are `[sp, sm, uz, lz, wc]` and the conceptual model unpacks them + positionally, so a short list is an `IndexError` inside the per-cell loop and a + long one silently ignores the extras. + """ + model = Catchment("coello", "2009-01-01", "2009-01-10") + + with pytest.raises(ValueError, match="state variables are 5"): + model.read_lumped_model(HBVLumped, 1530, initial_cond) + + def test_a_parameter_set_reports_how_many_parameters_it_carries(self): + """Test that `count` reports the width the set was checked against. + + Test scenario: + The width rule is enforced on construction, so `count` is how a caller reads + back what it settled on -- `12` for the no-snow, no-MAXBAS configuration. + """ + parameters = ParameterSet(np.ones(12), snow=False, maxbas=False) + + assert parameters.count == 12, ( + f"a 12-value set should report 12, got {parameters.count}" + ) + + def test_bounds_of_different_lengths_are_refused(self): + """Test that a lower and upper bound of different lengths cannot pair up. + + Test scenario: + The two are read from separate files, so they can disagree. The optimiser + samples between them per position, and a mismatch means positions with only one + side. + """ + with pytest.raises(ValueError, match="same as LB"): + ParameterBounds([0.0] * 12, [1.0] * 11) + + def test_a_span_that_runs_backwards_is_refused(self): + """Test that an end date before the start is named rather than silently empty. + + Test scenario: + A backwards span produces an empty `date_index`, which surfaces much later as a + zero-length driver mismatch naming neither date. + """ + with pytest.raises(ValueError, match="ends before it starts"): + SimulationPeriod.parse("2009-12-31", "2009-01-01") + def test_the_calendar_is_built_once(self): """Test that the derived calendar is cached rather than rebuilt on every read. diff --git a/tests/rrm/catchment/test_results.py b/tests/rrm/catchment/test_results.py index b2d0774e..b057a962 100644 --- a/tests/rrm/catchment/test_results.py +++ b/tests/rrm/catchment/test_results.py @@ -19,7 +19,7 @@ import pytest from hapi.catchment import Catchment -from hapi.inputs import FlowNetwork, MeteoInputs +from hapi.inputs import FlowNetwork, MeteoInputs, RiverGeometry from hapi.results import STATE_VARIABLES, RoutingKind, SimulationResults from hapi.routing import Routing from hapi.rrm.distrrm import DistributedRRM @@ -254,6 +254,52 @@ def test_route_maxbas_by_path_length_records_itself( assert results.q_total is not None, "the per-cell fields must be filled" +class TestTheHydraulicCellSkip: + """The flood model's handoff: river cells this package deliberately does not route.""" + + def test_river_cells_are_left_unrouted_when_the_skip_is_asked_for( + self, distributed_run: DistributedRun + ): + """Test that a positive bankfull depth keeps a cell out of the routing. + + Args: + distributed_run: The validated run, rebuilt here with a river geometry. + + Test scenario: + `skip_hydraulic_cells` is the handoff to a 1D hydraulic model, which routes + those cells instead. The branch that acts on it was never exercised: the flood + tests check that the *warning* fires, not that the cells come out unrouted. + An unrouted cell keeps its own `quz`, with nothing accumulated from upstream. + """ + model = distributed_run + rows, cols = model.flow_network.shape + depth = np.zeros((rows, cols)) + # A downstream cell, not a headwater: cells at accumulation 0 are copied straight + # across before the skip is consulted, so only a cell the second loop visits can + # show the branch doing anything. + downstream = model.flow_network.acc_val[-1] + river_x, river_y = model.flow_network.cells_by_acc_val[downstream][0] + depth[river_x, river_y] = 3.0 + flat = np.ones((rows, cols)) + with_geometry = DistributedRun( + period=model.period, + meteo=model.meteo, + flow_network=model.flow_network, + parameters=model.parameters, + model_setup=model.model_setup, + river_geometry=RiverGeometry(flat, depth, flat, flat, flat), + skip_hydraulic_cells=True, + ) + + results = DistributedRRM.run_lumped_model(with_geometry) + DistributedRRM.route_muskingum(with_geometry, results) + + assert np.array_equal( + results.quz_routed[river_x, river_y, :], + np.zeros(model.meteo.simulation_steps, dtype="float32"), + ), "a skipped river cell must be left for the hydraulic model, not routed here" + + class TestTheOutletShortcut: """Reading the outlet cell only means something for a scheme that accumulates.""" diff --git a/tests/rrm/catchment/test_run_narrowing.py b/tests/rrm/catchment/test_run_narrowing.py index 6c686dc2..43216640 100644 --- a/tests/rrm/catchment/test_run_narrowing.py +++ b/tests/rrm/catchment/test_run_narrowing.py @@ -210,6 +210,96 @@ def test_geometry_off_the_catchment_grid_is_refused(self, built: Catchment): DistributedRun.from_model(built, with_river_geometry=True) +class TestTheInvariantsHoldWhenTheRunIsBuiltDirectly: + """`from_model` is the front door, but `__post_init__` is what actually guarantees.""" + + def test_a_parameter_cube_of_the_wrong_width_is_refused(self, built: Catchment): + """Test that a parameter cube with too few columns is refused. + + Args: + built: A fully built distributed catchment. + + Test scenario: + The rows check has a test; the columns check did not, so half the guard was + unexercised. Both matter: the cube is indexed by `[x, y, :]` in the per-cell + loop, and a narrow cube reads a cell that belongs to a different column. + """ + cube = built.parameters.values + built.parameters = built.parameters.with_values(cube[:, :-1, :]) + + with pytest.raises(ValueError, match="columns"): + DistributedRun.from_model(built) + + def test_a_skip_built_directly_is_refused_too(self, built: Catchment): + """Test that the skip guard holds on the constructor, not only on `from_model`. + + Args: + built: A fully built distributed catchment. + + Test scenario: + `from_model` refuses this earlier with a message naming `read_river_geometry`, + so the constructor's own guard never ran in the suite. It is the one that + actually holds, because a run can be built without going through `from_model`. + """ + run = DistributedRun.from_model(built) + + with pytest.raises(ValueError, match="skipping the hydraulic cells"): + DistributedRun( + period=run.period, + meteo=run.meteo, + flow_network=run.flow_network, + parameters=run.parameters, + model_setup=run.model_setup, + skip_hydraulic_cells=True, + ) + + def test_a_direction_raster_without_its_table_is_refused(self, built: Catchment): + """Test that a network carrying a raster but no lookup table is caught up front. + + Args: + built: A fully built distributed catchment. + + Test scenario: + `from_rasters` always derives the table alongside the raster, so the two + normally travel together -- but `FlowNetwork` can be constructed directly, and + the routing loop indexes the table for every cell. This is the check that keeps + the pair honest for a network built by hand. + """ + network = built.flow_network + built.flow_network = FlowNetwork( + network.flow_acc_arr, + no_data_value=network.no_data_value, + cell_size=network.cell_size, + px_area=network.px_area, + flow_dir_arr=network.flow_dir_arr, + ) + + with pytest.raises(ValueError, match="flow-direction table"): + DistributedRun.from_model(built, needs_flow_direction=True) + + def test_the_routing_table_says_when_the_network_has_none(self, built: Catchment): + """Test that asking for the routing table without a direction raster explains why. + + Args: + built: A fully built distributed catchment. + + Test scenario: + MAXBAS runs legitimately build a `FlowNetwork` with no direction raster, so a + run can exist without a table. Reaching `routing_table` on one used to be a + `KeyError` on a `None` dict inside the routing loop. + """ + built.flow_network = FlowNetwork( + built.flow_network.flow_acc_arr, + no_data_value=built.flow_network.no_data_value, + cell_size=built.flow_network.cell_size, + px_area=built.flow_network.px_area, + ) + run = DistributedRun.from_model(built, needs_flow_direction=False) + + with pytest.raises(ValueError, match="flow-direction table"): + _ = run.routing_table + + class TestTheEnginesCannotBeReachedUnvalidated: """The seam is enforced by the signatures, not by remembering to call it.""" diff --git a/tests/rrm/catchment/test_run_validation.py b/tests/rrm/catchment/test_run_validation.py index 0caf1c14..71e0bcb2 100644 --- a/tests/rrm/catchment/test_run_validation.py +++ b/tests/rrm/catchment/test_run_validation.py @@ -229,6 +229,28 @@ def test_rejects_a_lake_record_of_the_wrong_length( "the wrapper must not run against a mismatched lake record" ) + def test_rejects_a_lake_that_was_never_given_a_record( + self, coello_loaded: Catchment, spied_wrapper: dict + ): + """Test that a lake with no meteorological data names the reader that supplies it. + + Args: + coello_loaded: A distributed catchment with every input read. + spied_wrapper: Records whether an engine was reached. + + Test scenario: + `Lake.MeteoData` is `None` until `read_meteo_data` runs, and a lake-aware entry + point handed such a lake used to fail on `np.shape(None)[0]` -- an `IndexError` + naming neither the lake nor the call nobody made. + """ + lake = _LakeStub(coello_loaded.meteo.time_steps) + lake.MeteoData = None + + with pytest.raises(ValueError, match="read_meteo_data"): + Run.run_distributed_with_lake(coello_loaded, lake) + + assert not spied_wrapper, "the engine must not be reached without a lake record" + @pytest.mark.parametrize("columns", [2, 3]) def test_rejects_a_lake_record_missing_a_column( self, coello_loaded: Catchment, spied_wrapper: dict, columns: int diff --git a/tests/rrm/catchment/test_wrapper_with_lake.py b/tests/rrm/catchment/test_wrapper_with_lake.py index 4a55a79d..74312a62 100644 --- a/tests/rrm/catchment/test_wrapper_with_lake.py +++ b/tests/rrm/catchment/test_wrapper_with_lake.py @@ -507,6 +507,48 @@ def test_records_maxbas_on_the_results( ) +class TestTheLakeInputGuards: + """`_lake_inputs` names whichever of the three readers was not called.""" + + @pytest.mark.parametrize( + "missing, expected", + [ + ("MeteoData", "read_meteo_data"), + ("Parameters", "read_parameters"), + ("OutflowCell", "outflow_cell"), + ], + ) + def test_an_unread_lake_input_is_named( + self, + coello_with_lake_inputs: Catchment, + coello_start_date: str, + coello_end_date: str, + missing: str, + expected: str, + ): + """Test that each missing lake input is reported by the reader that supplies it. + + Args: + coello_with_lake_inputs: A distributed catchment ready for a lake run. + coello_start_date: Start of the simulation. + coello_end_date: End of the simulation. + missing: The lake attribute to clear. + expected: Substring naming the reader the error must point at. + + Test scenario: + All three are indexed straight by the wrapper -- `MeteoData[:, 0]`, + `Parameters[11]`, `OutflowCell[0]` -- so an unread one used to fail on `None` + several frames in, naming a subscript rather than the call nobody made. + """ + model = coello_with_lake_inputs + lake = _make_lake(model, coello_start_date, coello_end_date, seed=11) + setattr(lake, missing, None) + run = DistributedRun.from_model(model) + + with pytest.raises(ValueError, match=expected): + Wrapper.run_muskingum_with_lake(run, lake) + + class TestRunDistributedWithLakeEndToEnd: """Tests that the public entry point completes with the record it documents.""" From bc3e1b7f4d865a5e436f306345a9578e1b917e79 Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Thu, 10 Sep 2026 23:17:05 +0200 Subject: [PATCH 40/54] docs: document what this round's fixes changed, and put calibration under doctest `Catchment.extract_discharge` gained a guard this round -- it refuses results no routing step has filled -- and its `Raises:` section did not mention it. That is the one correctness item here: a documented contract that had stopped matching the code. The rest is the symbols whose *meaning* changed, each with an executable example rather than a description of one: * `SimulationResults.outlet_shortcut_valid` -- it now excludes `UNROUTED` as well as `MAXBAS`, so the example shows which kinds do and do not support reading the outlet cell. * `RoutingKind` -- the values a run records on its results. * `SimulationPeriod.date_index` -- cached now, so the example pins that the same object comes back; `freq` and `conversion_factor` got theirs alongside, the latter showing the factor of 24 between the resolutions. * `ParameterSet.count` -- reads back the width the set was checked against, which for a distributed cube is the parameters per cell, not the cells. * `ObjectiveFunctionArityError` -- new public name this round. Its example shows the property the design turns on: it is still a `ValueError`, so code already wrapping a calibration in `except ValueError` keeps catching it. `src/hapi/calibration.py` joins the `doctests` task, so those examples are checked rather than trusted. 46 doctests pass, up from 39; `mkdocs build --strict` stays clean. Not done, deliberately: `Run.*`, `Wrapper.*`, `DistributedRRM.*` and the `Calibration` entry points still carry no examples. Every one needs a raster dataset and a full run to demonstrate, which is why their modules sit outside the doctest task -- an example there would be prose that drifts. They are covered by the test suite instead. --- pyproject.toml | 2 +- src/hapi/calibration.py | 25 +++++++++++++++++++ src/hapi/catchment.py | 4 ++- src/hapi/conceptual.py | 26 ++++++++++++++++++- src/hapi/period.py | 55 +++++++++++++++++++++++++++++++++++++++-- src/hapi/results.py | 35 ++++++++++++++++++++++++++ 6 files changed, 142 insertions(+), 5 deletions(-) diff --git a/pyproject.toml b/pyproject.toml index 2786293f..98dc9946 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -265,7 +265,7 @@ cmd = [ "src/hapi/conceptual.py", "src/hapi/inputs.py", "src/hapi/period.py", "src/hapi/results.py", "src/hapi/runs.py", "src/hapi/routing.py", - "src/hapi/run.py", + "src/hapi/run.py", "src/hapi/calibration.py", ] description = "Run the doctests of the modules whose examples are executable" diff --git a/src/hapi/calibration.py b/src/hapi/calibration.py index d168d2cd..19a81c4f 100644 --- a/src/hapi/calibration.py +++ b/src/hapi/calibration.py @@ -40,6 +40,31 @@ class ObjectiveFunctionArityError(ValueError): was caught by that handler one line later and scored `np.nan`, so a caller who wired up an objective of the wrong arity got a full Harmony Search over an all-`nan` landscape and a warning per trial instead of the error this message was written for. + + It stays a `ValueError` subclass, so code that already wraps a calibration in + `except ValueError` keeps catching it. + + Examples: + - It carries the message that names what the objective needs: + ```python + >>> from hapi.calibration import ( + ... OBJECTIVE_FN_ARGS_ERROR, + ... ObjectiveFunctionArityError, + ... ) + >>> str(ObjectiveFunctionArityError(OBJECTIVE_FN_ARGS_ERROR))[:24] + 'the objective function y' + + ``` + - An existing `ValueError` handler still catches it: + ```python + >>> from hapi.calibration import ObjectiveFunctionArityError + >>> try: + ... raise ObjectiveFunctionArityError("needs more inputs") + ... except ValueError as exc: + ... print(exc) + needs more inputs + + ``` """ diff --git a/src/hapi/catchment.py b/src/hapi/catchment.py index 24803d45..19941a8e 100644 --- a/src/hapi/catchment.py +++ b/src/hapi/catchment.py @@ -1046,7 +1046,9 @@ def extract_discharge(self, calculate_metrics=True, factor=None): per-gauge (Muskingum) path. Default is None. Raises: - ValueError: The gauge table has not been read, or the model has not been run. + ValueError: The gauge table has not been read, the model has not been run, or + the results it produced have not been routed -- there is no hydrograph to + extract from a set of arrays no routing step has filled. """ if self.GaugesTable is None: raise ValueError("please read the gauges' table first.") diff --git a/src/hapi/conceptual.py b/src/hapi/conceptual.py index 2488a122..cb749641 100644 --- a/src/hapi/conceptual.py +++ b/src/hapi/conceptual.py @@ -198,7 +198,31 @@ def __post_init__(self): @property def count(self) -> int: - """int: Number of parameters the set carries. See :func:`parameter_count`.""" + """int: Number of parameters the set carries. See :func:`parameter_count`. + + The width rule is enforced when the set is built, so this reads back what it + settled on -- for a distributed set, the length of the trailing axis rather than + the number of cells. + + Examples: + - A lumped set is a flat vector, so the count is its length: + ```python + >>> import numpy as np + >>> from hapi.conceptual import ParameterSet + >>> ParameterSet(np.ones(12), snow=False, maxbas=False).count + 12 + + ``` + - A distributed set counts the parameters per cell, not the cells: + ```python + >>> import numpy as np + >>> from hapi.conceptual import ParameterSet + >>> cube = np.ones((13, 14, 12)) + >>> ParameterSet(cube, snow=False, maxbas=False).count + 12 + + ``` + """ return parameter_count(self.values) def with_values(self, values: np.ndarray | list) -> ParameterSet: diff --git a/src/hapi/period.py b/src/hapi/period.py index affbc56f..6356bf9c 100644 --- a/src/hapi/period.py +++ b/src/hapi/period.py @@ -146,7 +146,21 @@ def parse( @property def freq(self) -> str: - """str: The pandas offset alias for this resolution.""" + """str: The pandas offset alias for this resolution. + + Examples: + - Each supported resolution maps to the alias `pd.date_range` expects: + ```python + >>> from hapi.period import SimulationPeriod + >>> SimulationPeriod.parse("2009-01-01", "2009-01-10").freq + 'D' + >>> SimulationPeriod.parse( + ... "2009-01-01", "2009-01-02", temporal_resolution="Hourly" + ... ).freq + 'h' + + ``` + """ return RESOLUTIONS[self.temporal_resolution] @cached_property @@ -158,6 +172,24 @@ def date_index(self) -> pd.DatetimeIndex: than rebuilt, which the frozen class makes safe -- the inputs it derives from cannot change, so the cache cannot go stale. It is read once per `from_model` (so once per calibration trial) and twice per `SimulationResults._step_bounds` call. + + Examples: + - One entry per step, inclusive of both ends: + ```python + >>> from hapi.period import SimulationPeriod + >>> period = SimulationPeriod.parse("2009-01-01", "2009-01-05") + >>> [step.strftime("%m-%d") for step in period.date_index] + ['01-01', '01-02', '01-03', '01-04', '01-05'] + + ``` + - Built once and handed back, because the span it describes cannot change: + ```python + >>> from hapi.period import SimulationPeriod + >>> period = SimulationPeriod.parse("2009-01-01", "2009-12-31") + >>> period.date_index is period.date_index + True + + ``` """ return pd.date_range(self.start, self.end, freq=self.freq) @@ -168,7 +200,26 @@ def days(self) -> int: @property def conversion_factor(self) -> float: - """float: Depth-to-discharge factor -- mm over the catchment to m3/s at this step.""" + """float: Depth-to-discharge factor -- mm over the catchment to m3/s at this step. + + It is the number of seconds in a step divided by 1000, so an hourly step is a + twenty-fourth of a daily one. + + Examples: + - The two resolutions differ by exactly a factor of 24: + ```python + >>> from hapi.period import SimulationPeriod + >>> daily = SimulationPeriod.parse("2009-01-01", "2009-01-10") + >>> hourly = SimulationPeriod.parse( + ... "2009-01-01", "2009-01-02", temporal_resolution="Hourly" + ... ) + >>> daily.conversion_factor + 86.4 + >>> round(daily.conversion_factor / hourly.conversion_factor, 1) + 24.0 + + ``` + """ return ( CONVERSION_FACTOR if self.temporal_resolution == "daily" diff --git a/src/hapi/results.py b/src/hapi/results.py index d0618eba..c10943d5 100644 --- a/src/hapi/results.py +++ b/src/hapi/results.py @@ -96,6 +96,17 @@ class RoutingKind(Enum): MUSKINGUM: Cell-to-cell Muskingum routing along the flow network. MAXBAS: Triangular (MAXBAS) routing of each cell straight to the outlet. LUMPED: No spatial routing -- the catchment was run as a single unit. + + Examples: + - The kind carries its own name, which is what a run records on its results: + ```python + >>> from hapi.results import RoutingKind + >>> RoutingKind.MUSKINGUM.value + 'muskingum' + >>> sorted(kind.value for kind in RoutingKind) + ['lumped', 'maxbas', 'muskingum', 'unrouted'] + + ``` """ UNROUTED = "unrouted" @@ -215,6 +226,30 @@ def outlet_shortcut_valid(self) -> bool: a contribution rather than a discharge -- reading the outlet cell of a MAXBAS run under-reports the hydrograph, which is what this guards. False for UNROUTED too: there is no `q_total` yet, so there is no cell to read and no shortcut to take. + + Examples: + - Muskingum accumulates downstream, so a cell is a discharge: + ```python + >>> import numpy as np + >>> from hapi.results import RoutingKind, SimulationResults + >>> cube = np.zeros((2, 3, 4), dtype="float32") + >>> muskingum = SimulationResults(RoutingKind.MUSKINGUM, cube, cube, None) + >>> muskingum.outlet_shortcut_valid + True + + ``` + - MAXBAS and unrouted arrays do not support the shortcut: + ```python + >>> import numpy as np + >>> from hapi.results import RoutingKind, SimulationResults + >>> cube = np.zeros((2, 3, 4), dtype="float32") + >>> [ + ... SimulationResults(kind, cube, cube, None).outlet_shortcut_valid + ... for kind in (RoutingKind.MAXBAS, RoutingKind.UNROUTED) + ... ] + [False, False] + + ``` """ return self.routing not in (RoutingKind.MAXBAS, RoutingKind.UNROUTED) From bd4c978e42513cbcf6c76d199fef238652106cb8 Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Thu, 10 Sep 2026 23:48:19 +0200 Subject: [PATCH 41/54] fix(conceptual): stop holding the search space to the conceptual model's width Round 1 added `validate_parameter_count` to `ParameterBounds.__post_init__`. That was wrong, and it broke every distributed calibration. A `ParameterSet` and a `ParameterBounds` bound two different things. A parameter *set* is what the conceptual model reads: 12 values per cell for `(snow=False, maxbas=False)`. The *bounds* delimit the flat vector the optimiser searches, whose length is the spatial distribution's `ParametersNO` -- `no_elem * no_parameters (+ no_lumped_par)`. On the shipped Coello grid that is **980** totally distributed and **243** for HRUs. All three were refused. There was no width that worked: pass 12 to satisfy the new check and `par3d` then raises `length of input parameters should be 89*(12 - 1) + 1 = 891`. The two coincide only for a lumped calibration, where the trial vector *is* the parameter set -- so the rule moves to `calibrate_lumped`, before the optimisation problem is declared, which is where round 1's concern (a mismatch surfacing once per trial from inside the objective) is actually true. Nothing caught this because every distributed-calibration test hand-assigns `ParameterBounds(np.zeros(12), np.ones(12))` rather than a realistic search space, and round 1's own test parametrised three *lumped* widths. It now asserts the opposite for the distributed case -- 12, 243 and 980 are all accepted -- keeps the length-mismatch rule, which holds whatever the search width is, and covers the lumped check on the path where it belongs. --- src/hapi/calibration.py | 12 ++- src/hapi/conceptual.py | 19 +++-- tests/rrm/calibration/test_rrm_calibration.py | 84 +++++++++++++++---- 3 files changed, 89 insertions(+), 26 deletions(-) diff --git a/src/hapi/calibration.py b/src/hapi/calibration.py index 19a81c4f..5f88867b 100644 --- a/src/hapi/calibration.py +++ b/src/hapi/calibration.py @@ -17,7 +17,7 @@ from Oasis.optimization import Optimization from hapi.catchment import Catchment -from hapi.conceptual import ParameterBounds, ParameterSet +from hapi.conceptual import ParameterBounds, ParameterSet, validate_parameter_count from hapi.inputs import MeteoInputs from hapi.protocols import SpatialDistribution from hapi.results import SimulationResults @@ -787,6 +787,16 @@ def calibrate_lumped( f"{', '.join(missing)} is missing" ) + # A lumped calibration is the one case where the optimiser's search vector *is* the + # parameter set the conceptual model reads, so its width has to match. The rule + # cannot live on `ParameterBounds`: a distributed calibration searches + # `SpatialVarFun.ParametersNO` values -- 980 on the shipped Coello grid -- and + # mapping them onto the grid is the whole job of the spatial distribution. + lumped_bounds = self._search_space() + validate_parameter_count( + lumped_bounds.lower, lumped_bounds.snow, lumped_bounds.maxbas + ) + route = basic_inputs["Route"] routing_fn = basic_inputs["RoutingFn"] if "InitialValues" in basic_inputs: diff --git a/src/hapi/conceptual.py b/src/hapi/conceptual.py index cb749641..f5a63cf1 100644 --- a/src/hapi/conceptual.py +++ b/src/hapi/conceptual.py @@ -293,8 +293,8 @@ class ParameterBounds: produces is checked against it. Attributes: - lower: Lower bound per parameter. - upper: Upper bound per parameter. + lower: Lower bound per element of the optimiser's search vector. + upper: Upper bound per element of the same vector. snow: Whether the snow routine runs. maxbas: Whether the parameter vector carries a MAXBAS value. @@ -317,19 +317,20 @@ def __post_init__(self): """Check the two bounds describe the same parameters. Raises: - ValueError: The bounds are different lengths, or do not hold the number of - parameters `(snow, maxbas)` calls for. + ValueError: The bounds are different lengths. """ if len(self.lower) != len(self.upper): raise ValueError( f"the length of UB should be the same as LB, got {len(self.upper)} and " f"{len(self.lower)}" ) - # The same rule every trial vector is held to. Checked here as well because this is - # where the configuration enters: a mismatch used to surface once per trial from - # `ParameterSet`, after the whole optimisation problem had been declared and the - # optimiser started, rather than at the call that got it wrong. - validate_parameter_count(self.lower, self.snow, self.maxbas) + # No width rule here, deliberately. These bounds delimit the *optimiser's* flat + # search vector, whose length is the spatial distribution's `ParametersNO` -- + # `no_elem * no_parameters (+ no_lumped_par)`, 980 for a totally distributed Coello + # run and 243 for the HRU one. A `ParameterSet` is a different thing: the parameters + # the conceptual model reads, 12 per cell. The two coincide only for a lumped + # calibration, where the trial vector *is* the parameter set, and + # `Calibration.calibrate_lumped` checks it there. object.__setattr__(self, "lower", np.array(self.lower)) object.__setattr__(self, "upper", np.array(self.upper)) diff --git a/tests/rrm/calibration/test_rrm_calibration.py b/tests/rrm/calibration/test_rrm_calibration.py index 928291d0..69999e32 100644 --- a/tests/rrm/calibration/test_rrm_calibration.py +++ b/tests/rrm/calibration/test_rrm_calibration.py @@ -25,32 +25,84 @@ def test_read_parameters_bounds( assert isinstance(Coello.bounds.maxbas, bool) -@pytest.mark.parametrize( - "width, snow, maxbas", - [(10, False, False), (12, True, True), (16, False, True)], -) -def test_read_parameters_bounds_refuses_a_width_the_configuration_does_not_call_for( - coello_rrm_date: list, width: int, snow: bool, maxbas: bool +@pytest.mark.parametrize("width", [12, 243, 980]) +def test_read_parameters_bounds_accepts_any_search_width( + coello_rrm_date: list, width: int ): - """Test that bounds of the wrong width are refused where they enter. + """Test that the bounds are not held to the conceptual model's parameter count. Args: coello_rrm_date: Start and end dates for the model. width: Number of bound values supplied. - snow: Whether the snow routine is on. - maxbas: Whether MAXBAS routing is on. Test scenario: - `ParameterSet` holds every trial vector to `PARAMETER_COUNTS[(snow, maxbas)]`, but - nothing held the *bounds* to it -- so a mismatch surfaced once per trial, from inside - the objective, after the whole optimisation problem had been declared and the - optimiser started. The bounds are where the configuration enters; the rule belongs - there too. + `ParameterBounds` delimits the *optimiser's* flat search vector, whose length is the + spatial distribution's `ParametersNO` -- 980 for a totally distributed run on the + Coello grid and 243 for the HRU one, against 12 for a lumped one. Holding it to + `PARAMETER_COUNTS` made every distributed calibration raise at + `read_parameters_bound`, and no width satisfied both that rule and `par3d`'s. """ coello = Calibration(Catchment("rrm", coello_rrm_date[0], coello_rrm_date[1])) - with pytest.raises(ValueError): - coello.read_parameters_bound([0.0] * width, [1.0] * width, snow, maxbas=maxbas) + coello.read_parameters_bound([0.0] * width, [1.0] * width, False) + + assert len(coello.bounds) == width, ( + f"the search space is {width} wide; got {len(coello.bounds)}" + ) + + +def test_read_parameters_bounds_still_refuses_mismatched_lengths( + coello_rrm_date: list, +): + """Test that a lower and upper bound of different lengths are still refused. + + Args: + coello_rrm_date: Start and end dates for the model. + + Test scenario: + The two are read from separate files and the optimiser samples between them per + position, so this rule holds whatever the search width is -- it is the one thing + `ParameterBounds` can check without knowing how the vector is mapped onto the grid. + """ + coello = Calibration(Catchment("rrm", coello_rrm_date[0], coello_rrm_date[1])) + + with pytest.raises(ValueError, match="same as LB"): + coello.read_parameters_bound([0.0] * 12, [1.0] * 11, False) + + +@pytest.mark.parametrize("width", [10, 16]) +def test_a_lumped_calibration_checks_the_search_width_before_it_starts( + coello_rrm_date: list, + lumped_meteo_data_path: str, + coello_AreaCoeff: float, + coello_InitialCond: list, + width: int, +): + """Test that the lumped path holds the search vector to the model's parameter count. + + Args: + coello_rrm_date: Start and end dates for the model. + lumped_meteo_data_path: Driver record. + coello_AreaCoeff: Catchment area. + coello_InitialCond: Initial state. + width: A search width the conceptual model cannot read. + + Test scenario: + A lumped calibration is the one case where the optimiser's vector *is* the parameter + set, so a mismatch there is a real error -- and it used to surface once per trial + from inside the objective, after the optimiser had started. Checked before the + problem is declared, where the caller can act on it. + """ + coello = Calibration(Catchment("rrm", coello_rrm_date[0], coello_rrm_date[1])) + coello.model.read_lumped_inputs(lumped_meteo_data_path) + coello.model.read_lumped_model(HBVLumped, coello_AreaCoeff, coello_InitialCond) + coello.read_parameters_bound([0.0] * width, [1.0] * width, False) + coello.read_objective_function(metrics.rmse, []) + + with pytest.raises(ValueError, match="takes 12 parameters"): + coello.calibrate_lumped( + dict(Route=0, RoutingFn=None), [{}, None, {}], print_error=None + ) def test_lumped_calibration( From c7720fcbde174c36d918d331bcdd5c4bdd28819f Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Tue, 15 Sep 2026 19:47:51 +0200 Subject: [PATCH 42/54] fix(calibration): read the objective's arity from its signature, not from a TypeError Round 1 made the "objective needs more inputs" error escape the handler that scores a trial infeasible. The diagnosis behind it is a blanket `except TypeError` wrapped around the objective call *and* the Muskingum constraint loop after it. While the error it raised was swallowed one line later that over-catch was harmless -- a misclassified trial was still just a `nan`. Once it escaped, any `TypeError` raised for a *value* reason inside a correctly wired objective ended the whole search and blamed the signature. The objective is user-supplied and is handed two pandas frames, so a `TypeError` there is not exotic. Arity is a property of the wiring, not of a trial: it is knowable before the search starts. `_check_objective_arity` binds the objective's signature against the number of arguments its entry point passes, and each of the three calls it once, up front. A wrongly wired objective is now reported before a single trial runs -- earlier than round 1 managed -- and the per-trial handler goes back to treating every runtime failure as one bad candidate. The check immediately found a wrongly wired test: `TestCalibrateMaxbas` passed `metrics.rmse`, which takes two arguments, while `calibrate_maxbas` calls `objective(QGauges, qout, GaugesTable)`. Every trial in that test raised `TypeError` and scored `nan`; it passed anyway because it only asserted on the stubbed optimiser's canned result. It uses a three-argument objective now. Two tests hold the two halves apart: an objective of the wrong arity reaches the caller, and a `TypeError` raised on the values is still just an infeasible trial with the search continuing. --- src/hapi/calibration.py | 114 ++++++++++-------- .../test_calibration_distributed.py | 59 ++++++++- 2 files changed, 123 insertions(+), 50 deletions(-) diff --git a/src/hapi/calibration.py b/src/hapi/calibration.py index 5f88867b..0aa45aa1 100644 --- a/src/hapi/calibration.py +++ b/src/hapi/calibration.py @@ -8,6 +8,7 @@ from __future__ import annotations +import inspect from collections.abc import Callable from typing import Any @@ -276,6 +277,39 @@ def _check_before_optimising(self, **narrowing: Any) -> None: DistributedRun.from_model(self.model, **narrowing) self._objective() + def _check_objective_arity(self, arguments: int) -> None: + """Check the objective can be called the way this entry point calls it. + + Read off the signature rather than by calling it: arity is a property of the wiring, + knowable before a single trial runs. It used to be diagnosed from a `TypeError` + raised *by* the call, which cannot tell "you gave me too few arguments" apart from a + `TypeError` raised inside a correctly-wired objective for a value reason -- and the + `try` covered the Muskingum constraint loop after the call as well. Classifying that + as a wiring error would end the whole search and blame the signature; classifying it + as a bad candidate, which is what happens now, is right. + + Args: + arguments: How many positional arguments this entry point passes, `of_args` + included. + + Raises: + ObjectiveFunctionArityError: The objective cannot accept that many. + """ + objective, of_args = self._objective() + try: + signature = inspect.signature(objective) + except (TypeError, ValueError): + # A builtin or C function with no introspectable signature. Nothing to check. + return + try: + signature.bind(*([None] * arguments)) + except TypeError as exc: + raise ObjectiveFunctionArityError( + f"{OBJECTIVE_FN_ARGS_ERROR}; this entry point passes {arguments} " + f"({arguments - len(of_args)} of its own plus {len(of_args)} from " + f"read_objective_function)" + ) from exc + def _search_space(self) -> ParameterBounds: """Return the bounds, or say which reader supplies them. @@ -499,6 +533,8 @@ def run_calibration( _check_optimization_args(api_obj_args, api_solve_args) self._check_before_optimising() + # `objective(QGauges, GaugesTable)` -- the shape this entry point calls with. + self._check_objective_arity(2) print("Calibration starts") ### calculate the objective function @@ -518,21 +554,16 @@ def opt_fun(par): try: self.model.results = Wrapper.run_muskingum(run) # calculate performance of the model - try: - error = objective( - self.model.QGauges, *[self.model.GaugesTable] - ) # self.model.results.qout, self.model.results.quz_routed, self.model.results.qlz_translated, - f = list(range(9, len(par), spatial_var_fun.no_parameters)) - g = list() - for i in range(len(f)): - k = par[f[i]] - x = par[f[i] + 1] - g.append(2 * k * x / self.model.period.dt) - g.append((2 * k * (1 - x)) / self.model.period.dt) - - except TypeError as e: - # the objective function received fewer inputs than it needs - raise ObjectiveFunctionArityError(OBJECTIVE_FN_ARGS_ERROR) from e + error = objective( + self.model.QGauges, *[self.model.GaugesTable] + ) # self.model.results.qout, self.model.results.quz_routed, self.model.results.qlz_translated, + f = list(range(9, len(par), spatial_var_fun.no_parameters)) + g = list() + for i in range(len(f)): + k = par[f[i]] + x = par[f[i] + 1] + g.append(2 * k * x / self.model.period.dt) + g.append((2 * k * (1 - x)) / self.model.period.dt) # print error if print_error != 0: @@ -540,10 +571,6 @@ def opt_fun(par): print(par) fail = 0 - except ObjectiveFunctionArityError: - # Not a bad parameter set: the objective function itself is wired up wrong, - # and every trial would fail the same way. Let it out. - raise except Exception as exc: # A genuine numerical failure for this candidate. Narrowed from a bare # `except`, which also caught KeyboardInterrupt -- so a long calibration @@ -655,6 +682,8 @@ def calibrate_maxbas( _check_optimization_args(api_obj_args, api_solve_args) self._check_before_optimising(needs_flow_direction=False) + # `objective(QGauges, qout, GaugesTable)`. + self._check_objective_arity(3) print("Calibration starts") # calculate the objective function @@ -672,25 +701,17 @@ def opt_fun(par): try: self.model.results = Wrapper.run_maxbas(run) # calculate performance of the model - try: - error = objective( - self.model.QGauges, - self.model.results.qout, - *[self.model.GaugesTable], - ) - except TypeError as e: - # the objective function received fewer inputs than it needs - raise ObjectiveFunctionArityError(OBJECTIVE_FN_ARGS_ERROR) from e - + error = objective( + self.model.QGauges, + self.model.results.qout, + *[self.model.GaugesTable], + ) # print error if print_error != 0: print(round(error, 3)) print(par) fail = 0 - except ObjectiveFunctionArityError: - # See run_calibration: a wrongly-wired objective is not a bad candidate. - raise except Exception as exc: # See run_calibration: narrowed from a bare `except`. logger.warning(f"trial failed, scoring it infeasible: {exc!r}") @@ -792,6 +813,9 @@ def calibrate_lumped( # cannot live on `ParameterBounds`: a distributed calibration searches # `SpatialVarFun.ParametersNO` values -- 980 on the shipped Coello grid -- and # mapping them onto the grid is the whole job of the spatial distribution. + # `objective(observed, Qsim, *of_args)`. + self._check_objective_arity(2 + len(self._objective()[1])) + lumped_bounds = self._search_space() validate_parameter_count( lumped_bounds.lower, lumped_bounds.snow, lumped_bounds.maxbas @@ -835,29 +859,21 @@ def opt_fun(par): self.model.results = run_results self.Qsim = run_results.q_total # calculate performance of the model - try: - error = objective( - observed[observed.columns[-1]], - self.Qsim, - *of_args, - ) - g = [ - 2 * par[-2] * par[-1] / self.model.period.dt, - (2 * par[-2] * (1 - par[-1])) / self.model.period.dt, - ] - except TypeError as e: - # the objective function received fewer inputs than it needs - raise ObjectiveFunctionArityError(OBJECTIVE_FN_ARGS_ERROR) from e - + error = objective( + observed[observed.columns[-1]], + self.Qsim, + *of_args, + ) + g = [ + 2 * par[-2] * par[-1] / self.model.period.dt, + (2 * par[-2] * (1 - par[-1])) / self.model.period.dt, + ] if print_error != 0: print( f"Error = {round(error, 3)} Inequality Const = {np.round(g, 2)}" ) # print(par) fail = 0 - except ObjectiveFunctionArityError: - # See run_calibration: a wrongly-wired objective is not a bad candidate. - raise except Exception as exc: # A genuine numerical failure for this candidate. Narrowed from a bare # `except`, which also caught KeyboardInterrupt -- so a long calibration diff --git a/tests/rrm/calibration/test_calibration_distributed.py b/tests/rrm/calibration/test_calibration_distributed.py index 650fb65a..e4de4c18 100644 --- a/tests/rrm/calibration/test_calibration_distributed.py +++ b/tests/rrm/calibration/test_calibration_distributed.py @@ -255,6 +255,36 @@ def needs_four_arguments(observed, simulated, gauges, extra): with pytest.raises(ObjectiveFunctionArityError, match="needs more inputs"): coello.run_calibration(spatial_var_stub, _optimization_args()) + def test_a_type_error_from_inside_a_correct_objective_is_not_a_wiring_error( + self, gauged_calibration: Calibration, stub_optimizer: dict, spatial_var_stub + ): + """Test that a `TypeError` raised for a value reason does not abort the search. + + Test scenario: + The arity diagnosis used to be a blanket `except TypeError` around the objective + call *and* the Muskingum constraint loop after it. While the error it raised was + swallowed one line later that over-catch was harmless, but once it escaped, any + `TypeError` from inside a correctly-wired objective ended the whole calibration + and blamed the signature. A `TypeError` there is not exotic: the objective is + user-supplied and is handed two pandas frames. Arity is read off the signature + before the search starts, so this one is just a bad candidate. + """ + coello = gauged_calibration + coello.bounds = ParameterBounds(np.zeros(12), np.ones(12)) + + def right_arity_wrong_values(qgauges, gauges_table): + """Take the two arguments the call site passes, then fail on the values.""" + return "a string" + 1 + + coello.read_objective_function(right_arity_wrong_values, []) + + coello.run_calibration(spatial_var_stub, _optimization_args()) + + assert "n_vars" in stub_optimizer, ( + "a TypeError on the values is one bad candidate, not a wiring error; the " + "optimiser should still have been driven" + ) + def test_a_numerically_failing_trial_is_still_scored_infeasible( self, gauged_calibration: Calibration, stub_optimizer: dict, spatial_var_stub ): @@ -410,7 +440,12 @@ def test_stores_the_optimizer_result_on_the_instance( separate code path that also had to be rewired onto `flow_network`. """ coello = gauged_calibration - coello.read_objective_function(metrics.rmse, []) + # Three arguments, because `calibrate_maxbas` calls + # `objective(QGauges, qout, GaugesTable)`. It used to be wired with `metrics.rmse`, + # which takes two -- so every trial raised `TypeError`, was scored `nan`, and this + # test still passed because it only ever asserted on the stubbed optimiser's canned + # result. The arity check refuses that wiring up front now. + coello.read_objective_function(_outlet_objective, []) coello.bounds = ParameterBounds(np.zeros(12), np.ones(12)) res = coello.calibrate_maxbas(spatial_var_stub, _optimization_args()) @@ -622,6 +657,28 @@ def test_a_mismatched_initial_values_length_is_refused( ) +def _outlet_objective(qgauges, qout, gauges_table) -> float: + """Score a trial the way `calibrate_maxbas` actually calls the objective. + + That entry point invokes `objective_function(QGauges, results.qout, GaugesTable)`, so an + objective taking two arguments cannot be called at all -- every trial raised `TypeError` + and was scored infeasible. This has the arity the call site uses, so the body runs. + + Args: + qgauges: Observed discharge frame. + qout: The outlet hydrograph the run produced. + gauges_table: Gauge metadata frame. + + Returns: + float: A finite score derived from all three. + """ + return float( + np.abs(qgauges.to_numpy(dtype=float)).mean() + + float(np.nansum(qout)) + + len(gauges_table) + ) + + def _pairwise_objective(qgauges, gauges_table) -> float: """Score a trial the way `run_calibration` actually calls the objective. From 7b48c6e59dd035e6187765f31dc91a290a62405a Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Tue, 15 Sep 2026 19:50:19 +0200 Subject: [PATCH 43/54] fix(examples): call the spatial distribution the way its signature reads Round 1 diagnosed this correctly -- `SpatialVarFun.Function` wants the optimiser's flat vector, not a `ParameterSet` -- and then wrote the fix into one of three sibling call sites, leaving the keywords wrong even there. `Function` is one of `par3d` / `par3d_lumped` / `hydrologic_response_units` / `par2d_lumped_k1_lake`, and every one of them takes exactly `(self, par_g)`. The `kub`/`klb` keywords were commented out of those signatures years ago, so all three calls raised `TypeError: par3d() got an unexpected keyword argument 'kub'` -- including the one round 1 "fixed". The other two also passed the wrong object: a `ParameterSet` in the totally-distributed script, and its `(13, 14, 12)` cube in the `tests/` one, both products of the mechanical `Coello.parameters` -> `Coello.model.parameters` rewrite. All three pass `Coello.best_parameters` now, with no keywords. The same two scripts also built the distribution with `DP(..., klb=, kub=)`, whose parameters are `k_lower_bound` / `k_upper_bound` -- so they raised before reaching the call above. Verified against the installed class: the constructor takes the long names, `Function(flat_vector)` returns a `(13, 14, 12)` `Par3d`, and `Function(..., kub=, klb=)` still refuses. --- ...rus-distributed-model-calibratation-muskingum.py | 10 ++++------ ...lly-distributed-model-calibratation-muskingum.py | 13 +++++++------ tests/calibration/distributed_mode_calib.py | 13 +++++++------ 3 files changed, 18 insertions(+), 18 deletions(-) diff --git a/examples/hydrological-model/coello/calibration/coello-hrus-distributed-model-calibratation-muskingum.py b/examples/hydrological-model/coello/calibration/coello-hrus-distributed-model-calibratation-muskingum.py index c8aa7cd8..37cd258d 100644 --- a/examples/hydrological-model/coello/calibration/coello-hrus-distributed-model-calibratation-muskingum.py +++ b/examples/hydrological-model/coello/calibration/coello-hrus-distributed-model-calibratation-muskingum.py @@ -139,10 +139,8 @@ def objective_function(q_obs, coordinates): # Qout, q_uz_routed, q_lz_trans, # %% run calibration cal_parameters = Coello.run_calibration(SpatialVarFun, OptimizationArgs, print_error=0) # %% convert parameters to rasters -# `best_parameters` is the flat vector the optimiser produced, which is what this -# function maps onto the grid. `model.parameters` is a `ParameterSet` -- a different -# shape describing a different thing. -SpatialVarFun.Function( - Coello.best_parameters, kub=SpatialVarFun.Kub, klb=SpatialVarFun.Klb -) +# `Function` takes exactly one argument, the flat vector the optimiser produced -- +# `par3d(self, par_g)`. `kub`/`klb` were commented out of the signature years ago, and +# `model.parameters` is a `ParameterSet`, a different shape describing a different thing. +SpatialVarFun.Function(Coello.best_parameters) SpatialVarFun.save_parameters(SaveTo) diff --git a/examples/hydrological-model/coello/calibration/coello-totally-distributed-model-calibratation-muskingum.py b/examples/hydrological-model/coello/calibration/coello-totally-distributed-model-calibratation-muskingum.py index 098f1e4f..60752ca7 100644 --- a/examples/hydrological-model/coello/calibration/coello-totally-distributed-model-calibratation-muskingum.py +++ b/examples/hydrological-model/coello/calibration/coello-totally-distributed-model-calibratation-muskingum.py @@ -53,7 +53,7 @@ UB & LB with the order of Klb then kub function inside the calibration algorithm is written as following -par_dist = SpatialVarFun(par,*SpatialVarArgs,kub=kub,klb=klb) +par_dist = SpatialVarFun(par) """ raster = Dataset.read_file(FlowAccPath) # ------------- @@ -71,8 +71,8 @@ no_lumped_par=no_lumped_par, lumped_par_pos=lumped_par_pos, function=2, - klb=klb, - kub=kub, + k_lower_bound=klb, + k_upper_bound=kub, ) # calculate no of parameters that optimization algorithm is going to generate print(SpatialVarFun.ParametersNO) @@ -137,7 +137,8 @@ def objective_function(q_obs, coordinates): # %% ### Run Calibration cal_parameters = Coello.run_calibration(SpatialVarFun, OptimizationArgs, print_error=1) # %% -SpatialVarFun.Function( - Coello.model.parameters, kub=SpatialVarFun.Kub, klb=SpatialVarFun.Klb -) +# `Function` takes exactly one argument, the flat vector the optimiser produced -- +# `par3d(self, par_g)`. `kub`/`klb` were commented out of the signature years ago, and +# `model.parameters` is a `ParameterSet`, a different shape describing a different thing. +SpatialVarFun.Function(Coello.best_parameters) SpatialVarFun.save_parameters(SaveTo) diff --git a/tests/calibration/distributed_mode_calib.py b/tests/calibration/distributed_mode_calib.py index 03a02335..9987bbdd 100644 --- a/tests/calibration/distributed_mode_calib.py +++ b/tests/calibration/distributed_mode_calib.py @@ -57,7 +57,7 @@ for muskingum parameters k & x include the upper and lower bound in both UB & LB with the order of Klb then kub function inside the calibration algorithm is written as following -par_dist=spatial_var_fun(par,*SpatialVarArgs,kub=kub,klb=klb) +par_dist = spatial_var_fun(par) """ raster = Dataset.read_file(FlowAccPath) @@ -76,8 +76,8 @@ no_lumped_par=no_lumped_par, lumped_par_pos=lumped_par_pos, function=2, - klb=klb, - kub=kub, + k_lower_bound=klb, + k_upper_bound=kub, ) # calculate no of parameters that optimization algorithm is going to generate spatial_var_fun.ParametersNO @@ -143,7 +143,8 @@ def objective_function(Qobs, Qout, q_uz_routed, q_lz_trans, coordinates): # [0.700, 399, 1.704, 0.1021, 0.4622, 0.6237, 0.1251, 0.005, 59.85, 5.241, # 94.91, 0.2075] # ) -spatial_var_fun.Function( - Coello.model.parameters.values, kub=spatial_var_fun.Kub, klb=spatial_var_fun.Klb -) +# `Function` takes exactly one argument, the flat vector the optimiser produced -- +# `par3d(self, par_g)`. `kub`/`klb` were commented out of the signature years ago, and +# `model.parameters` is a `ParameterSet`, a different shape describing a different thing. +spatial_var_fun.Function(Coello.best_parameters) spatial_var_fun.save_parameters(SaveTo) From 8b6938ed8131b1464f88ca3869b5d5eb1723840c Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Tue, 15 Sep 2026 19:52:32 +0200 Subject: [PATCH 44/54] fix(catchment): name the missing outlet series instead of reshaping None Round 1 made the routers record their own routing kind and made `extract_discharge` refuse results nothing had routed. Between those two it opened a gap: `route_maxbas_by_path_length` now labels its results `RoutingKind.MAXBAS`, but unlike `route_maxbas` it has no `Wrapper` entry point to sum the domain after it, so those results arrive labelled MAXBAS with `qout` still empty. That is the one state the new guard cannot see -- it only tests for `UNROUTED`. `np.reshape(None, n)` then reported `cannot reshape array of size 1 into shape (10,)`, naming neither the field nor the step that should have filled it. Before round 1 the same call said `'NoneType' object is not subscriptable`; the fix relabelled the failure rather than closing it. The MAXBAS branch requires `qout` by name now, and the message says where it comes from: the `Wrapper` entry points fill it, because summing the domain is not a routing step. `_record_maxbas`'s docstring says the same, so a caller driving `DistributedRRM` directly knows what they still owe. The round-1 test asserted the routing kind and `q_total` and stopped one field short; it now also pins that `qout` is deliberately empty, and a second test walks the whole path-length route through `extract_discharge`. --- src/hapi/catchment.py | 11 ++++++-- src/hapi/results.py | 4 ++- src/hapi/rrm/distrrm.py | 5 ++++ tests/rrm/catchment/test_results.py | 41 +++++++++++++++++++++++++++++ 4 files changed, 58 insertions(+), 3 deletions(-) diff --git a/src/hapi/catchment.py b/src/hapi/catchment.py index 19941a8e..4dc148d5 100644 --- a/src/hapi/catchment.py +++ b/src/hapi/catchment.py @@ -1124,10 +1124,17 @@ def extract_discharge(self, calculate_metrics=True, factor=None): ) else: # MAXBAS: a cell of `q_total` is a contribution, so the hydrograph is the - # basin-wide sum the run already put in `qout`. + # basin-wide sum the run already put in `qout`. Required by name rather than + # reshaped straight: `DistributedRRM.route_maxbas_by_path_length` records the + # routing but has no wrapper to sum the domain after it, so its results reach + # here labelled MAXBAS with `qout` still empty -- and `np.reshape(None, n)` + # reports "cannot reshape array of size 1", naming neither the field nor the + # step that should have filled it. self.Qsim = pd.DataFrame(index=self.period.date_index) gauge_id = self.GaugesTable.loc[self.GaugesTable.index[-1], "id"] - q_sim = np.reshape(self.results.qout, self.meteo.time_steps) + q_sim = np.reshape( + self.results._require_field("qout"), self.meteo.time_steps + ) self.Qsim.loc[:, gauge_id] = q_sim if calculate_metrics: diff --git a/src/hapi/results.py b/src/hapi/results.py index c10943d5..e28f4601 100644 --- a/src/hapi/results.py +++ b/src/hapi/results.py @@ -330,7 +330,9 @@ def _require_field(self, name: str) -> np.ndarray: if value is None: raise ValueError( f"`{name}` is empty because no routing step has filled it; these results " - f"are {self.routing.value}" + f"are {self.routing.value}. `qout` is filled by the `Wrapper` entry points, " + f"not by the routers -- summing the domain is not a routing step -- so " + f"results routed by calling `DistributedRRM` directly do not carry one" ) return value diff --git a/src/hapi/rrm/distrrm.py b/src/hapi/rrm/distrrm.py index 55498390..9f89a444 100644 --- a/src/hapi/rrm/distrrm.py +++ b/src/hapi/rrm/distrrm.py @@ -213,6 +213,11 @@ def _record_maxbas(results: SimulationResults) -> None: writes through the alias -- but the alias is visible (`results.quz_routed is results.quz`), so an in-place edit of one changes the other. + It does not fill `qout`. The outlet hydrograph is the sum over the domain, which is + not a routing step -- the `Wrapper` entry points do it after calling a router. So + results produced by driving `DistributedRRM` directly carry `RoutingKind.MAXBAS` + and no `qout`, and `extract_discharge` says so by name. + Args: results: The results whose `quz` / `qlz` have just been routed. Mutated in place. """ diff --git a/tests/rrm/catchment/test_results.py b/tests/rrm/catchment/test_results.py index b057a962..02b6dc41 100644 --- a/tests/rrm/catchment/test_results.py +++ b/tests/rrm/catchment/test_results.py @@ -252,6 +252,10 @@ def test_route_maxbas_by_path_length_records_itself( f"the path-length router must record what it applied, got {results.routing}" ) assert results.q_total is not None, "the per-cell fields must be filled" + assert results.qout is None, ( + "summing the domain is not a routing step, so a router leaves `qout` empty; " + "the `Wrapper` entry points are what fill it" + ) class TestTheHydraulicCellSkip: @@ -333,6 +337,43 @@ def test_the_shortcut_is_valid_only_where_a_cell_is_a_discharge( f"{routing} should give outlet_shortcut_valid={valid}" ) + def test_extracting_maxbas_results_without_an_outlet_series_names_the_field( + self, built_catchment: Catchment, maxbas_run: DistributedRun + ): + """Test that MAXBAS results carrying no `qout` say which field is missing. + + Args: + built_catchment: A distributed catchment with its gauge table read. + maxbas_run: A validated run carrying a MAXBAS parameter set. + + Test scenario: + `route_maxbas_by_path_length` records the routing but has no wrapper to sum the + domain after it, so its results reach `extract_discharge` labelled MAXBAS with + `qout` still empty -- the one state the UNROUTED guard cannot see, because it + only tests for UNROUTED. `np.reshape(None, n)` then reported "cannot reshape + array of size 1", naming neither the field nor the step that should have filled + it. + """ + rows, cols = maxbas_run.flow_network.shape + gradient = np.arange(rows * cols, dtype=float).reshape(rows, cols) + gradient[np.isnan(maxbas_run.flow_network.flow_acc_arr)] = np.nan + with_fpl = DistributedRun( + period=maxbas_run.period, + meteo=maxbas_run.meteo, + flow_network=maxbas_run.flow_network, + parameters=maxbas_run.parameters, + model_setup=maxbas_run.model_setup, + flow_path_length=gradient, + ) + results = DistributedRRM.run_lumped_model(with_fpl) + DistributedRRM.route_maxbas_by_path_length(with_fpl, results) + + catchment = copy(built_catchment) + catchment.results = results + + with pytest.raises(ValueError, match="`qout` is empty"): + catchment.extract_discharge() + def test_extracting_from_unrouted_results_names_the_missing_step( self, built_catchment: Catchment ): From 0e0ed9f3fa9a11b827ff4f879389a4777156ba4d Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Tue, 15 Sep 2026 19:54:23 +0200 Subject: [PATCH 45/54] docs: make the two run walkthroughs match the API they document Round 1 fixed `run-configuration.md`'s argument order but paired `Routing.triangular_routing_1` with `coello-lumped-model-run.yaml`, whose parameter set declares `maxbas: false`. `Wrapper.run_lumped` picks the routing signature off that flag, so the snippet took the Muskingum branch and called a two-argument function with five: `TypeError: triangular_routing_1() takes 2 positional arguments but 5 were given`. It uses `Routing.muskingum_v` now, matching the shipped script beside it, and says which config to switch to for the triangular function. Executed from the repo root: 1,095 steps against a 1,095-step period. The snippet also loaded the config by a bare filename while every script on the branch uses the repo-root path; it now does too. `distributed-model-run.md` is the one example page the diff never touched, which is why both rounds fixed the four that are in it and left this one. It still built `Calibration(name, Sdate, Edate)` -- a signature that now raises `TypeError` -- and called five readers, `GaugesTable`, `extract_discharge` and `Qsim` on the `Calibration` rather than on `.model`. Four more things were wrong with it independently of that: `gdal.Open` where `Parameters` requires a pyramids `Dataset`; `Function=`/`Klb=`/`Kub=` where the constructor takes `function=`/`k_lower_bound=`/ `k_upper_bound=`; `SpatialVarFun.Function(Coello.parameters, kub=..., klb=...)`, which is the same call H2 corrected in three scripts; and an objective declared with five parameters where `run_calibration` passes two -- which, since this round made arity a checked property, is now reported up front instead of scoring `nan`. Its `read_objective_function` call was indented inside the function body, so it never ran at all. --- docs/examples/distributed-model-run.md | 54 +++++++++++++++----------- docs/examples/run-configuration.md | 11 +++++- 2 files changed, 41 insertions(+), 24 deletions(-) diff --git a/docs/examples/distributed-model-run.md b/docs/examples/distributed-model-run.md index 59f8ecc3..4a63d936 100644 --- a/docs/examples/distributed-model-run.md +++ b/docs/examples/distributed-model-run.md @@ -13,6 +13,7 @@ import numpy as np import datetime as dt from osgeo import gdal from hapi.calibration import Calibration +from hapi.catchment import Catchment from hapi.inputs import FlowNetwork, MeteoInputs from hapi.rrm.hbv_bergestrom92 import HBVBergestrom92 as HBV @@ -38,20 +39,22 @@ Snow = 0 Sdate = '2009-01-01' Edate = '2011-12-31' name = "Coello" -Coello = Calibration(name, Sdate, Edate, spatial_resolution="Distributed") +# `Calibration` holds a catchment rather than being one, so the model is built first +# and every reader is called on `Coello.model`. +Coello = Calibration(Catchment(name, Sdate, Edate, spatial_resolution="Distributed")) # Meteorological & GIS Data -Coello.meteo = MeteoInputs.from_rasters(PrecPath, TempPath, Evap_Path) +Coello.model.meteo = MeteoInputs.from_rasters(PrecPath, TempPath, Evap_Path) -Coello.flow_network = FlowNetwork.from_rasters(FlowAccPath, FlowDPath) +Coello.model.flow_network = FlowNetwork.from_rasters(FlowAccPath, FlowDPath) # Lumped Model -Coello.read_lumped_model(HBV, AreaCoeff, InitialCond) +Coello.model.read_lumped_model(HBV, AreaCoeff, InitialCond) # Gauges Data -Coello.read_gauge_table(Path + "/stations/gauges.csv", FlowAccPath) +Coello.model.read_gauge_table(Path + "/stations/gauges.csv", FlowAccPath) GaugesPath = Path + "/stations/" -Coello.read_discharge_gauges(GaugesPath, column='id', fmt="%Y-%m-%d") +Coello.model.read_discharge_gauges(GaugesPath, column='id', fmt="%Y-%m-%d") @@ -62,8 +65,10 @@ Coello.read_discharge_gauges(GaugesPath, column='id', fmt="%Y-%m-%d") ```python from hapi.rrm.parameters import Parameters as DP +from pyramids.dataset import Dataset -raster = gdal.Open(FlowAccPath) +# A pyramids `Dataset`, not a bare GDAL handle — `Parameters.__init__` refuses anything else. +raster = Dataset.read_file(FlowAccPath) #------------- # for lumped catchment parameters no_parameters = 12 @@ -74,7 +79,8 @@ no_lumped_par = 1 lumped_par_pos = [7] SpatialVarFun = DP(raster, no_parameters, no_lumped_par=no_lumped_par, - lumped_par_pos=lumped_par_pos,Function=2, Klb=klb, Kub=kub) + lumped_par_pos=lumped_par_pos, function=2, + k_lower_bound=klb, k_upper_bound=kub) # calculate no of parameters that optimization algorithm is going to generate SpatialVarFun.ParametersNO @@ -84,22 +90,24 @@ SpatialVarFun.ParametersNO ```python -coordinates = Coello.GaugesTable[['id','x','y','weight']][:] +coordinates = Coello.model.GaugesTable[['id','x','y','weight']][:] -# define the objective function and its arguments -OF_args = [coordinates] - -def objective_function(Qobs, Qout, q_uz_routed, q_lz_trans, coordinates): - Coello.extract_discharge() - all_errors=[] +# `run_calibration` calls the objective as `objective(QGauges, GaugesTable)` — two +# positional arguments, and nothing else. The arity is checked against the signature +# before the search starts, so a mismatch is reported rather than scored as `nan`. +def objective_function(Qobs, gauges_table): + Coello.model.extract_discharge() + all_errors = [] # error for all internal stations - for i in range(len(coordinates)): - all_errors.append((metrics.rmse(Qobs.loc[:,Qobs.columns[0]],Coello.Qsim[:,i]))) #*coordinates.loc[coordinates.index[i],'weight'] - print(all_errors) - error = sum(all_errors) - return error + for i in range(len(gauges_table)): + all_errors.append( + metrics.rmse(Qobs.loc[:, Qobs.columns[0]], Coello.model.Qsim.iloc[:, i]) + ) + return sum(all_errors) + - Coello.read_objective_function(objective_function, OF_args) +# Registered outside the function — indented inside it, this line never ran. +Coello.read_objective_function(objective_function, []) ``` ## Calibration algorithm Arguments @@ -133,6 +141,8 @@ cal_parameters = Coello.run_calibration(SpatialVarFun, OptimizationArgs,print_er ## Save results ```python -SpatialVarFun.Function(Coello.parameters, kub=SpatialVarFun.Kub, klb=SpatialVarFun.Klb) +# `best_parameters` is the flat vector the optimiser produced, which is what `Function` +# maps onto the grid; it takes that one argument and nothing else. +SpatialVarFun.Function(Coello.best_parameters) SpatialVarFun.save_parameters(SaveTo) ``` diff --git a/docs/examples/run-configuration.md b/docs/examples/run-configuration.md index ae3a9fa4..d231f967 100644 --- a/docs/examples/run-configuration.md +++ b/docs/examples/run-configuration.md @@ -14,10 +14,17 @@ from hapi.catchment import Catchment from hapi.routing import Routing from hapi.run import Run -Coello = Catchment.from_yaml("coello-lumped-model-run.yaml") +Coello = Catchment.from_yaml( + "examples/hydrological-model/coello/run/coello-lumped-model-run.yaml" +) # `Route` is the flag; the routing function is the third argument. Passing the function # as the flag routes with nothing, because a callable is truthy. -Run.run_lumped(Coello, 1, Routing.triangular_routing_1) +# +# The function has to match the parameter set the config declares. This one says +# `maxbas: false`, so `Wrapper.run_lumped` takes the Muskingum branch and calls the +# function with five arguments — a triangular function takes two and raises. Use +# `coello-lumped-model-run-maxbas.yaml` with `Routing.triangular_routing_1` instead. +Run.run_lumped(Coello, 1, Routing.muskingum_v) ``` The four shipped examples under `examples/hydrological-model/coello/run/` are each a pair — a From 611fd41a6d0f3ba55985db41b16b2e6ac9bf0906 Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Tue, 15 Sep 2026 19:58:17 +0200 Subject: [PATCH 46/54] fix: close the three gaps round 1's own fixes left behind Three small defects, each introduced or left open by a round-1 fix. `_save_rasters` guarded against handing the raster writer a view of the result arrays with `np.ascontiguousarray`. That returns its input untouched when the input is already contiguous, and numpy ignores size-1 dimensions when testing contiguity -- so a single-step range, which `save(start=d, end=d)` asks for, still passed a view. Measured: a one-step slice was not copied, two and five were. It uses `np.array(..., order="C", copy=True)`, and a test covers both the case that regressed and the case that always worked. `Calibration.extract_discharge` is a second implementation of the same method and did not get the `UNROUTED` refusal `Catchment.extract_discharge` gained. It read only `outlet_shortcut_valid`, which the same round-1 change widened to exclude `UNROUTED` -- so unrouted results reached a message stating categorically that the run used triangular routing. A confident wrong diagnosis is worse than the `AttributeError` it replaced. It makes both refusals now, in the same order, and a test asserts the message does not mention MAXBAS. The Jiboa example passed a directory-plus-prefix as `path`. `save` joins rather than concatenates now, so instead of prefixing the file names it created a directory literally named `Lumped_Parameters__` and wrote `Result_*.tif` inside it. The branch rewrote that call without updating the value it passes, while the four Coello scripts and both docs pages moved to the directory + `prefix=` form. --- .../Jiboa-distributed-model-muskingum-lake.py | 12 +++- src/hapi/calibration.py | 15 +++- src/hapi/results.py | 8 ++- .../test_calibration_distributed.py | 27 +++++++ .../test_save_results_distributed.py | 70 +++++++++++++++++++ 5 files changed, 125 insertions(+), 7 deletions(-) diff --git a/examples/hydrological-model/jiboa/Jiboa-distributed-model-muskingum-lake.py b/examples/hydrological-model/jiboa/Jiboa-distributed-model-muskingum-lake.py index 2fa2aa04..2cebde36 100644 --- a/examples/hydrological-model/jiboa/Jiboa-distributed-model-muskingum-lake.py +++ b/examples/hydrological-model/jiboa/Jiboa-distributed-model-muskingum-lake.py @@ -209,7 +209,15 @@ start_date = "2012-07-20" end_date = "2012-08-20" -Path = save_to + "Lumped_Parameters_" + str(dt.datetime.now())[0:10] + "_" +# `path` is the directory and `prefix` names the files. They used to be one +# concatenated string, which `save` now joins rather than concatenates -- so this wrote +# `Result_*.tif` into a directory literally called `Lumped_Parameters__`. +prefix = "Lumped_Parameters_" + str(dt.datetime.now())[0:10] + "_" Jiboa.results.save( - result=1, start=start_date, end=end_date, path=Path, flow_acc_path=flow_acc_path + result=1, + start=start_date, + end=end_date, + path=save_to, + prefix=prefix, + flow_acc_path=flow_acc_path, ) diff --git a/src/hapi/calibration.py b/src/hapi/calibration.py index 0aa45aa1..6c9bd352 100644 --- a/src/hapi/calibration.py +++ b/src/hapi/calibration.py @@ -21,7 +21,7 @@ from hapi.conceptual import ParameterBounds, ParameterSet, validate_parameter_count from hapi.inputs import MeteoInputs from hapi.protocols import SpatialDistribution -from hapi.results import SimulationResults +from hapi.results import RoutingKind, SimulationResults from hapi.runs import DistributedRun, LumpedRun from hapi.wrapper import Wrapper @@ -423,10 +423,19 @@ def extract_discharge( scaling is applied. Default is None. Raises: - ValueError: The results came from MAXBAS routing, whose per-cell values are - contributions rather than discharges. + ValueError: The results have not been routed, or came from MAXBAS routing, + whose per-cell values are contributions rather than discharges. """ results, meteo, gauges = self._gauged_results() + # The same two refusals `Catchment.extract_discharge` makes, in the same order. + # This one only tested `outlet_shortcut_valid`, which now excludes `UNROUTED` as + # well as `MAXBAS` -- so unrouted results reached a message stating categorically + # that the run used triangular routing, which it had not. + if results.routing is RoutingKind.UNROUTED: + raise ValueError( + "these results have not been routed, so there is no hydrograph to extract; " + "call a Run.* entry point rather than DistributedRRM.run_lumped_model alone" + ) if results.q_total is None: raise ValueError( "the results carry no routed discharge; the run did not complete" diff --git a/src/hapi/results.py b/src/hapi/results.py index e28f4601..c17bf093 100644 --- a/src/hapi/results.py +++ b/src/hapi/results.py @@ -751,8 +751,12 @@ def _save_rasters( cube = Datacube.from_dataset(src, arr.shape[2]) # A copy, not the `moveaxis` view: `arr` is a slice of a result array, and # handing a view to a writer that may normalise no-data in place would edit - # the results this call is only supposed to read. - cube.values = np.ascontiguousarray(np.moveaxis(arr, -1, 0)) + # the results this call is only supposed to read. `np.array(copy=True)` rather + # than `ascontiguousarray`, which returns the input untouched when it is + # already contiguous -- and a single-step range is, because numpy ignores + # size-1 dimensions when testing that. `save(start=d, end=d)` therefore kept + # handing out a view, which is exactly the case this guards. + cube.values = np.array(np.moveaxis(arr, -1, 0), order="C", copy=True) cube.to_file(names) def _save_csv( diff --git a/tests/rrm/calibration/test_calibration_distributed.py b/tests/rrm/calibration/test_calibration_distributed.py index e4de4c18..0fa9c62d 100644 --- a/tests/rrm/calibration/test_calibration_distributed.py +++ b/tests/rrm/calibration/test_calibration_distributed.py @@ -19,7 +19,9 @@ from hapi.inputs import FlowNetwork, MeteoInputs from hapi.results import RoutingKind, SimulationResults from hapi.routing import Routing +from hapi.rrm.distrrm import DistributedRRM from hapi.rrm.hbv_bergestrom92 import HBVBergestrom92 as HBVLumped +from hapi.runs import DistributedRun CANNED_RESULT = (0.42, np.arange(12, dtype="float64")) @@ -335,6 +337,31 @@ def test_a_trial_parameter_set_does_not_alias_the_optimizer_buffer( "the parameter set must not follow the buffer the next trial overwrites" ) + def test_extracting_unrouted_results_is_not_blamed_on_maxbas( + self, gauged_calibration: Calibration + ): + """Test that `Calibration.extract_discharge` names the missing routing step. + + Test scenario: + `Catchment.extract_discharge` gained an explicit `UNROUTED` refusal; this + parallel implementation was left reading only `outlet_shortcut_valid`, which the + same change widened to exclude `UNROUTED`. Unrouted results therefore reached a + message stating categorically that the run used triangular routing -- a + confident, wrong diagnosis. Two implementations of one method disagreeing about + one guard is what `RoutingKind` exists to prevent. + """ + coello = gauged_calibration + coello.model.results = DistributedRRM.run_lumped_model( + DistributedRun.from_model(coello.model) + ) + + with pytest.raises(ValueError, match="have not been routed") as exc: + coello.extract_discharge() + + assert "MAXBAS" not in str(exc.value), ( + f"unrouted results must not be blamed on MAXBAS: {exc.value}" + ) + def test_rejects_meteo_that_does_not_cover_the_grid( self, gauged_calibration: Calibration, stub_optimizer: dict, spatial_var_stub ): diff --git a/tests/rrm/catchment/test_save_results_distributed.py b/tests/rrm/catchment/test_save_results_distributed.py index 0271699b..d0674d10 100644 --- a/tests/rrm/catchment/test_save_results_distributed.py +++ b/tests/rrm/catchment/test_save_results_distributed.py @@ -12,6 +12,7 @@ import pytest from pyramids.dataset import Dataset +import hapi.results as results_module from hapi.catchment import Catchment from hapi.inputs import FlowNetwork, MeteoInputs from hapi.rrm.hbv_bergestrom92 import HBVBergestrom92 as HBVLumped @@ -246,6 +247,75 @@ def test_save_uses_the_prefix_it_is_given( ) +@pytest.mark.parametrize("end, steps", [("2009-01-01", 1), ("2009-01-03", 3)]) +def test_save_hands_the_writer_a_copy_not_a_view( + coello_run: Catchment, + coello_acc_path: str, + tmp_path, + monkeypatch, + end: str, + steps: int, +): + """Test that the array given to the raster writer never aliases the results. + + Args: + coello_run: Distributed Coello catchment with a completed run. + coello_acc_path: Path to the flow-accumulation raster used as the template. + tmp_path: Destination directory. + monkeypatch: Used to capture what is assigned to the cube. + end: End date, giving a one-step range and a three-step one. + steps: How many rasters that range covers. + + Test scenario: + A writer that normalises no-data in place would edit the result arrays a *save* is + only supposed to read. The guard was `np.ascontiguousarray`, which hands back its + input untouched when it is already contiguous -- and numpy ignores size-1 + dimensions when testing that, so the single-step range `save(start=d, end=d)` asks + for still passed a view. One step is the case that regressed; three is the case + that always worked. + """ + captured = {} + + class _RecordingCube: + """Stand-in for pyramids' DatasetCollection that records what it is given.""" + + @classmethod + def from_dataset(cls, src, time_length): + """Build the recorder instead of an in-memory scaffold.""" + captured["time_length"] = time_length + return cls() + + @property + def values(self): + """Return what was assigned.""" + return captured.get("values") + + @values.setter + def values(self, array): + captured["values"] = array + + def to_file(self, names): + """Record the names instead of writing them.""" + captured["names"] = names + + monkeypatch.setattr(results_module, "Datacube", _RecordingCube) + + coello_run.results.save( + path=str(tmp_path), + flow_acc_path=coello_acc_path, + result=1, + start="2009-01-01", + end=end, + ) + + assert captured["time_length"] == steps, ( + f"expected {steps} steps, got {captured['time_length']}" + ) + assert not np.shares_memory(captured["values"], coello_run.results.q_total), ( + "the writer must receive a copy; a view lets it edit the results in place" + ) + + def test_save_without_a_template_raster_says_what_it_needs( coello_run: Catchment, tmp_path ): From 2a924543711a4d73989dca3884c7c187bb13c3b7 Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Tue, 15 Sep 2026 20:01:00 +0200 Subject: [PATCH 47/54] fix(calibration): forward the objective's extra arguments, and guard the path-length range `read_objective_function(fn, args)` documents `args` as "extra arguments forwarded to it", and the `_objective()` helper this branch introduced says the same in its own `Returns:`. Two of the three entry points bound them and then never passed them: `run_calibration` and `calibrate_maxbas` called the objective with a hard-coded argument list. A caller who registered extra arguments got no error and no effect, which is the silent kind of wrong -- and ruff cannot see it, because `F841` does not report an unused name from tuple unpacking. All three forward `*of_args` now, and the arity check counts them. `route_maxbas_by_path_length` normalises by `max - min` over the flow-path-length raster with nothing checking that range is non-zero. A constant raster made every cell's MAXBAS NaN, surfacing as "Maxbas value has to be at least 1, got nan" from inside `triangular_routing_2` -- several frames away, naming a parameter the caller never set. It now names the raster and why it cannot be scaled. --- src/hapi/calibration.py | 18 ++++++----- src/hapi/rrm/distrrm.py | 9 ++++++ .../test_calibration_distributed.py | 29 +++++++++++++++++ tests/rrm/catchment/test_results.py | 32 +++++++++++++++++++ 4 files changed, 80 insertions(+), 8 deletions(-) diff --git a/src/hapi/calibration.py b/src/hapi/calibration.py index 6c9bd352..63671dc2 100644 --- a/src/hapi/calibration.py +++ b/src/hapi/calibration.py @@ -542,8 +542,8 @@ def run_calibration( _check_optimization_args(api_obj_args, api_solve_args) self._check_before_optimising() - # `objective(QGauges, GaugesTable)` -- the shape this entry point calls with. - self._check_objective_arity(2) + # `objective(QGauges, GaugesTable, *of_args)` -- the shape this entry point uses. + self._check_objective_arity(2 + len(self._objective()[1])) print("Calibration starts") ### calculate the objective function @@ -563,9 +563,10 @@ def opt_fun(par): try: self.model.results = Wrapper.run_muskingum(run) # calculate performance of the model - error = objective( - self.model.QGauges, *[self.model.GaugesTable] - ) # self.model.results.qout, self.model.results.quz_routed, self.model.results.qlz_translated, + # `of_args` forwarded, as `read_objective_function` documents. Two of the + # three entry points used to bind them and never pass them, so a caller who + # supplied extra arguments got no error and no effect. + error = objective(self.model.QGauges, self.model.GaugesTable, *of_args) f = list(range(9, len(par), spatial_var_fun.no_parameters)) g = list() for i in range(len(f)): @@ -691,8 +692,8 @@ def calibrate_maxbas( _check_optimization_args(api_obj_args, api_solve_args) self._check_before_optimising(needs_flow_direction=False) - # `objective(QGauges, qout, GaugesTable)`. - self._check_objective_arity(3) + # `objective(QGauges, qout, GaugesTable, *of_args)`. + self._check_objective_arity(3 + len(self._objective()[1])) print("Calibration starts") # calculate the objective function @@ -713,7 +714,8 @@ def opt_fun(par): error = objective( self.model.QGauges, self.model.results.qout, - *[self.model.GaugesTable], + self.model.GaugesTable, + *of_args, ) # print error if print_error != 0: diff --git a/src/hapi/rrm/distrrm.py b/src/hapi/rrm/distrrm.py index 9f89a444..da3e5385 100644 --- a/src/hapi/rrm/distrrm.py +++ b/src/hapi/rrm/distrrm.py @@ -283,6 +283,15 @@ def route_maxbas_by_path_length( MaxFPL = np.nanmax(run.flow_path_length) MinFPL = np.nanmin(run.flow_path_length) + # A constant raster makes the normalisation below divide by zero, which produces + # NaN and then surfaces as "Maxbas value has to be at least 1, got nan" from inside + # `triangular_routing_2`, several frames from the raster that caused it. + if MaxFPL == MinFPL: + raise ValueError( + f"the flow-path-length raster is constant at {MinFPL}, so there is no " + f"range to scale MAXBAS along; this routing needs cells at different " + f"distances from the outlet" + ) # resize_fun = lambda x: np.round(((((x - min_dist)/(max_dist - min_dist))*(1*maxbas - 1)) + 1), 0) resize_fun = lambda g: ( (((g - MinFPL) / (MaxFPL - MinFPL)) * (1 * MAXBAS - 1)) + 1 diff --git a/tests/rrm/calibration/test_calibration_distributed.py b/tests/rrm/calibration/test_calibration_distributed.py index 0fa9c62d..9efabeba 100644 --- a/tests/rrm/calibration/test_calibration_distributed.py +++ b/tests/rrm/calibration/test_calibration_distributed.py @@ -337,6 +337,35 @@ def test_a_trial_parameter_set_does_not_alias_the_optimizer_buffer( "the parameter set must not follow the buffer the next trial overwrites" ) + def test_the_objective_receives_the_extra_arguments_it_was_registered_with( + self, gauged_calibration: Calibration, stub_optimizer: dict, spatial_var_stub + ): + """Test that `read_objective_function`'s extra arguments reach the objective. + + Test scenario: + `read_objective_function(fn, args)` documents `args` as "extra arguments + forwarded to it", and `_objective()` returns them saying the same. Two of the + three entry points bound them and never passed them, so a caller who supplied + them got no error and no effect -- the silent kind of wrong. + """ + coello = gauged_calibration + coello.bounds = ParameterBounds(np.zeros(12), np.ones(12)) + seen = {} + + def objective_with_extras(qgauges, gauges_table, weight, label): + """Take the two the entry point passes plus the two registered with it.""" + seen["weight"] = weight + seen["label"] = label + return float(np.abs(qgauges.to_numpy(dtype=float)).mean()) + + coello.read_objective_function(objective_with_extras, [0.5, "outlet"]) + + coello.run_calibration(spatial_var_stub, _optimization_args()) + + assert seen == {"weight": 0.5, "label": "outlet"}, ( + f"the registered arguments must reach the objective, got {seen}" + ) + def test_extracting_unrouted_results_is_not_blamed_on_maxbas( self, gauged_calibration: Calibration ): diff --git a/tests/rrm/catchment/test_results.py b/tests/rrm/catchment/test_results.py index 02b6dc41..654a9313 100644 --- a/tests/rrm/catchment/test_results.py +++ b/tests/rrm/catchment/test_results.py @@ -304,6 +304,38 @@ def test_river_cells_are_left_unrouted_when_the_skip_is_asked_for( ), "a skipped river cell must be left for the hydraulic model, not routed here" +class TestPathLengthRoutingNeedsARange: + """Scaling MAXBAS along a distance needs cells at different distances.""" + + def test_a_constant_raster_is_refused(self, maxbas_run: DistributedRun): + """Test that a flow-path-length raster with no range says so. + + Args: + maxbas_run: A validated run carrying a MAXBAS parameter set. + + Test scenario: + The normalisation divides by `max - min`. A constant raster makes that zero, so + every cell's MAXBAS became NaN and the failure surfaced as "Maxbas value has to + be at least 1, got nan" from inside `triangular_routing_2` -- several frames + from the raster that caused it, and naming a parameter the caller never set. + """ + rows, cols = maxbas_run.flow_network.shape + flat = np.full((rows, cols), 7.0) + flat[np.isnan(maxbas_run.flow_network.flow_acc_arr)] = np.nan + run = DistributedRun( + period=maxbas_run.period, + meteo=maxbas_run.meteo, + flow_network=maxbas_run.flow_network, + parameters=maxbas_run.parameters, + model_setup=maxbas_run.model_setup, + flow_path_length=flat, + ) + results = DistributedRRM.run_lumped_model(run) + + with pytest.raises(ValueError, match="constant at"): + DistributedRRM.route_maxbas_by_path_length(run, results) + + class TestTheOutletShortcut: """Reading the outlet cell only means something for a scheme that accumulates.""" From 2d19a871f23a885df6508cbfeef26430a324e7a4 Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Tue, 15 Sep 2026 20:04:05 +0200 Subject: [PATCH 48/54] docs: correct seven descriptions that contradict the code they describe The biggest one is mine from round 1. Three docstrings explained the length trim as dropping "the leading initial-state slot" -- `[:-1]` drops the *trailing* element. The conceptual model sizes its arrays `len(prec) + 1` and writes indices 0..n-1, so index 0 holds the initial state, 1..n-1 the simulated steps, and index n is never written. Verified on the shipped lumped configuration: `q_total[0]` equals the untrimmed `q[0]`, and the dropped value is the unwritten `0.0` at the end. The trim is right and the calendar alignment is right; the explanation was backwards, and it hid something worth knowing -- `Qsim.iloc[0]` and `qout[0]` are the warm-up state, not a simulated value, which matters when scoring the first step. `_check_lake_meteo` was raised to four columns this round and every surrounding docstring was left at three -- including the `Raises:` section of the function that changed, both lake entry points, and `Lake.read_meteo_data`. All four name the long-term average temperature now. `save`'s option list still called options 2 and 3 "Upper zone discharge" and "Lower zone discharge" while `_RASTER_OPTIONS` writes `quz_routed` and `qlz_translated`. The same drift was corrected in `animate`'s list this round and this one was missed, so the two lists described the same arrays differently. `docs/api/catchment.md` said `Catchment` *and `Calibration`* accept a routing method; `Calibration` takes a model and no routing method at all. It also justified the spelling check by a comparison the branch deleted -- the routing loop no longer compares against `"Muskingum"`. The page now says what actually reads the stored method: `Run.run_flood` deriving the river-cell skip from `"Kinematic"`, and the cross-check against `parameters.maxbas`. `README.md` said "there is no compatibility alias: the names above are the only ones" directly after listing the *removed* CamelCase names -- reading as if `Run.RunHapi` is what survives, in the one paragraph a downstream reader consults about the break. Two costs are now written down rather than left to be discovered: the per-trial parameter copy is a full cube kept alive by `results.run` (~96 MB per retained result on a 1000x1000 grid), and `SimulationPeriod.dt` notes that the lake paths pass `conversion_factor` into the same `muskingum_v` parameter the catchment paths pass `dt` to -- an 86.4x difference between two routings of the same kind, which is the physics question already filed as issue #218. --- README.md | 3 ++- docs/api/catchment.md | 17 +++++++++++------ src/hapi/calibration.py | 7 +++++++ src/hapi/catchment.py | 10 ++++++---- src/hapi/period.py | 10 ++++++++-- src/hapi/results.py | 16 +++++++++------- src/hapi/run.py | 15 +++++++++------ src/hapi/wrapper.py | 16 ++++++++++------ 8 files changed, 62 insertions(+), 32 deletions(-) diff --git a/README.md b/README.md index 99e6542c..98f94cf5 100644 --- a/README.md +++ b/README.md @@ -127,4 +127,5 @@ Quick start - class method/function: snake_case (get_file, read_config). They should have a verb in them, because they perform some action. The CamelCase entry points that survived from earlier releases (`Run.RunHapi`, `Wrapper.RRMModel` and the rest) -have been renamed to snake_case. There is no compatibility alias: the names above are the only ones. +have been renamed to snake_case — `Run.run_distributed`, `Wrapper.run_muskingum`, and so on. There is no +compatibility alias: the old CamelCase spellings are gone. diff --git a/docs/api/catchment.md b/docs/api/catchment.md index fbe35a2d..9b63ad5e 100644 --- a/docs/api/catchment.md +++ b/docs/api/catchment.md @@ -2,8 +2,8 @@ ## Routing methods -`Catchment` and `Calibration` accept exactly three routing methods, matched case-insensitively -and stored in the one spelling the internals compare against: +`Catchment` accepts exactly three routing methods, matched case-insensitively and stored in one +spelling: | Written as | Stored as | Routes | |---|---|---| @@ -11,11 +11,16 @@ and stored in the one spelling the internals compare against: | `maxbas` | `MAXBAS` | Every cell straight to the outlet through a triangular function. | | `kinematic` | `Kinematic` | The flood model's own path (`Run.run_flood`). | +`Calibration` does not take one at all: it holds a `Catchment`, and reads the method off the model +it was given. + Anything else raises a `ValueError` naming the three. Before this check the constructor stored -whatever string it was handed, so a run configured as `"Max_bas"` — or as a descriptive label -such as `"Muskingum-Cunge"` — was accepted and then silently routed with Muskingum, because -`distrrm.route_muskingum` compares against `"Muskingum"` exactly. Rejecting the spelling is what -makes that comparison trustworthy; a script passing a spelling outside the table has to be updated +whatever string it was handed, so a run configured as `"Max_bas"` — or as a descriptive label such +as `"Muskingum-Cunge"` — was accepted and then silently routed with Muskingum, because the routing +loop compared against `"Muskingum"` exactly. That comparison is gone: which router runs is decided +by the entry point you call, and the stored method is read by `Run.run_flood`, which derives +`skip_hydraulic_cells` from `"Kinematic"`, and by the cross-check against `parameters.maxbas`. One +spelling is what keeps both honest; a script passing a spelling outside the table has to be updated to one of the three. A YAML run configuration reaches only the first two: `kinematic` selects the flood model, which diff --git a/src/hapi/calibration.py b/src/hapi/calibration.py index 63671dc2..9ca6184c 100644 --- a/src/hapi/calibration.py +++ b/src/hapi/calibration.py @@ -238,6 +238,13 @@ def _parameter_set(self, values) -> ParameterSet: overwrites in place. `results.run` is documented as the inputs those arrays came from; without the copy it described whichever trial happened to run last. + The copy is a full `(rows, cols, n_parameters)` cube and stays alive as long as the + results object holding it does. On the Coello grid that is 13x14x12 floats; on a + 1000x1000 grid it is about 96 MB per retained result, in the same loop + :attr:`~hapi.runs.DistributedRun.keep_state_variables` exists to keep small. Nothing + bounds it, because correctness came first -- provenance that describes a different + trial is worse than provenance that costs memory. + Args: values: The trial parameter array or vector. diff --git a/src/hapi/catchment.py b/src/hapi/catchment.py index 4dc148d5..abb20ec8 100644 --- a/src/hapi/catchment.py +++ b/src/hapi/catchment.py @@ -1077,9 +1077,10 @@ def extract_discharge(self, calculate_metrics=True, factor=None): # Muskingum accumulates downstream, so the outlet cell of `q_total` is the # outlet hydrograph. The engine cannot set this itself: finding the outlet # needs the gauge table, which is an analysis input, not a run input. - # Trimmed like every other path: `q_total` carries the conceptual model's - # initial-state slot, so the untrimmed form made `qout` a step longer here than - # on the MAXBAS and lake paths, for a field documented as one hydrograph. + # Trimmed like every other path: the conceptual model allocates one slot more + # than it fills, so the untrimmed form made `qout` a step longer here than on + # the MAXBAS and lake paths, for a field documented as one hydrograph. `[:-1]` + # drops the unwritten trailing slot; index 0 stays the model's initial state. self.results.qout = self.results.q_total[outlet_x, outlet_y, :-1] for i in range(len(self.GaugesTable)): @@ -1361,7 +1362,8 @@ def read_meteo_data(self, path: str, fmt: str): Args: path (str): Path to the meteorological data CSV file. Columns must be in the order [date, rainfall, ET, - temperature]. + temperature, long-term average temperature]. The lake + wrappers read that fourth driver as column 3. fmt (str): Date format string used to parse the date index. """ diff --git a/src/hapi/period.py b/src/hapi/period.py index 6356bf9c..c49dab9f 100644 --- a/src/hapi/period.py +++ b/src/hapi/period.py @@ -232,8 +232,14 @@ def dt(self) -> float: One for both resolutions today. It is a property rather than a stored `1` so the Muskingum routing has a single place to read it from; whether an hourly run should - route with a different value is an open question, recorded in the planning notes - rather than silently answered here. + route with a different value is an open question, filed as issue #218 rather than + silently answered here. + + Note that the catchment routing reads this while the lake paths in + :mod:`hapi.wrapper` pass :attr:`conversion_factor` (86.4 daily) into the same `dt` + parameter of the same `Routing.muskingum_v`. Both predate this class and neither was + changed when it was extracted, so one of the two is routing on a time step 86.4x off + from the other. Which one is the same physics question as issue #218. """ return 1.0 diff --git a/src/hapi/results.py b/src/hapi/results.py index c17bf093..6580d2fa 100644 --- a/src/hapi/results.py +++ b/src/hapi/results.py @@ -145,8 +145,10 @@ class SimulationResults: q_total: `quz_routed + qlz_translated`. Read it through :attr:`outlet_shortcut_valid` rather than assuming what a cell means. qout: The outlet hydrograph, when the run computed one, and always `len(period)` - long -- the conceptual model's leading initial-state slot is dropped on every - path that fills it. The MAXBAS paths sum over the domain and set it directly; + long. The conceptual model allocates one slot more than it fills, and every + path that produces a series drops that unwritten trailing slot -- so index 0 + is the model's initial state, not a simulated step, which matters when scoring + the first value against an observation. The MAXBAS paths sum over the domain and set it directly; the Muskingum paths leave it `None` for :meth:`~hapi.catchment.Catchment.extract_discharge` to read off the outlet cell, which needs the gauge table the engine does not have. @@ -597,11 +599,11 @@ def save( Args: path: Output directory for a distributed run (created if it does not exist), or the CSV file itself for a lumped one. Default is "", the working directory. - result: What to write. Distributed: 1 - Total discharge, 2 - Upper zone - discharge, 3 - Lower zone discharge, 4 - Snow pack, 5 - Soil moisture, - 6 - Upper zone, 7 - Lower zone, 8 - Water content. Lumped: 1 - simulated - discharge, 2 - upper zone, 3 - lower zone, 4 - the five states, 5 - all of - them. Default is 1. + result: What to write. Distributed: 1 - Total discharge, 2 - Surface flow (the + routed upper zone), 3 - Ground water flow (the translated lower zone), + 4 - Snow pack, 5 - Soil moisture, 6 - Upper zone, 7 - Lower zone, + 8 - Water content. Lumped: 1 - simulated discharge, 2 - upper zone, + 3 - lower zone, 4 - the five states, 5 - all of them. Default is 1. start: Start of the output period. A string is parsed with `fmt`. If empty, the run's first step. end: End of the output period, inclusive. If empty, the run's last step. diff --git a/src/hapi/run.py b/src/hapi/run.py index f526cd39..99a0fa9b 100644 --- a/src/hapi/run.py +++ b/src/hapi/run.py @@ -43,8 +43,9 @@ def _check_lake_meteo(run: DistributedRun, lake: LakeType) -> None: Raises: ValueError: The lake has no meteorological record, the record is a different length - from the distributed drivers, or it carries fewer than the three columns the - lake model reads. + from the distributed drivers, or it carries fewer than the four columns the + lake model reads, including the long-term average temperature the wrappers + index as column 3. """ meteo_data = lake.MeteoData if meteo_data is None: @@ -253,8 +254,9 @@ def run_distributed_with_lake( model: The model to run. See :class:`DistributedModel` for what it must carry. lake: Lake object containing lake configuration and meteorological data. Must have a `MeteoData` attribute - with shape `(time_steps, >= 3)` where columns are - rain, ET, and temperature. + with shape `(time_steps, >= 4)` where columns are + rain, ET, temperature, and the long-term average + temperature the wrappers read as column 3. Returns: SimulationResults: The run's output, also assigned to `model.results`. @@ -320,8 +322,9 @@ def run_maxbas_with_lake(model: CatchmentLike, lake: LakeType) -> SimulationResu model: The model to run. See :class:`DistributedModel` for what it must carry. lake: Lake object containing lake configuration and meteorological data. Must have a `MeteoData` attribute - with shape `(time_steps, >= 3)` where columns are - rain, ET, and temperature. + with shape `(time_steps, >= 4)` where columns are + rain, ET, temperature, and the long-term average + temperature the wrappers read as column 3. Returns: SimulationResults: The run's output, also assigned to `model.results`. diff --git a/src/hapi/wrapper.py b/src/hapi/wrapper.py index e802eb0c..5791955c 100644 --- a/src/hapi/wrapper.py +++ b/src/hapi/wrapper.py @@ -384,12 +384,16 @@ def run_lumped( # The lumped total discharge is exactly what `q_total` means, so it goes there rather # than onto the catchment as `Qsim`. `Run.run_lumped` is what indexes it by the period # and puts the frame on the model -- so this engine writes nothing outside `results`. - # The conceptual model prepends an initial-state slot, so its series is one step - # longer than the period covers. Trimmed once, here, rather than inside each - # routing branch: both routed branches used to do it and the unrouted one did not, - # so `Run.run_lumped(model)` -- the entry point's own default, `Route=0` -- produced - # an `n + 1` series and then raised `Length of values (1096) does not match length - # of index (1095)` when the period indexed it. + # The conceptual model allocates one slot more than it fills: `simulate` sizes its + # arrays `len(prec) + 1` and writes indices 0..n-1, so index 0 carries the initial + # state, 1..n-1 the simulated steps, and index n is never written. `[:-1]` drops + # that unwritten trailing slot -- the leading initial-state one is kept, which is + # why `q_total[0]` is the warm-up value rather than a simulated step. + # + # Trimmed once, here, rather than inside each routing branch: both routed branches + # used to do it and the unrouted one did not, so `Run.run_lumped(model)` -- the + # entry point's own default, `Route=0` -- produced an `n + 1` series and then raised + # `Length of values (1096) does not match length of index (1095)`. q_total = (results.quz + results.qlz)[:-1] if Routing != 0 and run.parameters.maxbas: From 787320e27a3e5fb3eb0280406343ef92f41a4827 Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Tue, 15 Sep 2026 20:07:13 +0200 Subject: [PATCH 49/54] fix: correct the swapped bounds, and the two docs pages left on the old constructor `tests/sensitivity_analysis.py` read `UB` from `LB-1-Muskinguk.txt` and `LB` from `UB-1-Muskinguk.txt`, then handed both to `SA(parameters, LB, UB, ...)` -- so every sample was drawn from the range upside down. Pre-existing, but this branch rewrote the two lines below it and both `SA` arguments, so the swap was inside the diff being edited. `distributed-model-calib.md` quoted and then called the pre-rename constructor (`StartDate`, `EndDate`, `SpatialResolution`, `TemporalResolution`) and omitted `routing_method` entirely; it also named `1-statevariables` where the field is `state_variables`, and read its gauges from `Hapi/Data/00inputs/...`, a path that does not exist in the repository and that ignored the page's own `Path` variable. `lumped-model-run.md` imported the *module* `hapi.rrm.hbv_bergestrom92` and passed it to `read_lumped_model`, which guards on `inspect.isclass` and refuses a module; passed `Title=` where the keyword is `title`; and read `Coello.QGauges['q']` on a page that never calls `read_discharge_gauges`, so the attribute was `None`. Five tests added in round 1 landed inside `TestTemporalResolution` and none of them was about temporal resolution. They move into `TestSimulationPeriod` and `TestConceptualModelInputs`, which is where someone would look for them. --- docs/examples/distributed-model-calib.md | 14 +-- docs/examples/lumped-model-run.md | 8 +- .../test_read_parameters_validation.py | 104 ++++++++++-------- tests/sensitivity_analysis.py | 7 +- 4 files changed, 74 insertions(+), 59 deletions(-) diff --git a/docs/examples/distributed-model-calib.md b/docs/examples/distributed-model-calib.md index 50630b9f..11fde0cd 100644 --- a/docs/examples/distributed-model-calib.md +++ b/docs/examples/distributed-model-calib.md @@ -9,11 +9,11 @@ The calibration of the Distributed rainfall runoff model follows the same steps class Catchment: - def __init__(self, name, StartDate, EndDate, fmt="%Y-%m-%d", SpatialResolution = 'Lumped', - TemporalResolution = "Daily"): + def __init__(self, name, start_data, end, fmt="%Y-%m-%d", spatial_resolution="Lumped", + temporal_resolution="Daily", routing_method="Muskingum"): """ ============================================================================= - Catchment(name, StartDate, EndDate, fmt="%Y-%m-%d", SpatialResolution = 'Lumped', + Catchment(name, start_data, end, fmt="%Y-%m-%d", spatial_resolution="Lumped", TemporalResolution = "Daily") ============================================================================= Parameters @@ -42,7 +42,7 @@ start = "2009-01-01" end = "2011-12-31" name = "Coello" -Coello = Catchment(name, start, end, SpatialResolution = "Distributed") +Coello = Catchment(name, start, end, spatial_resolution="Distributed") ``` # Read Meteorological Inputs @@ -92,8 +92,8 @@ Coello.read_lumped_model(HBV, CatchmentArea, InitialCond) - to check the performance of the model we need to read the gauge hydrographs ```python -Coello.read_gauge_table("Hapi/Data/00inputs/Discharge/stations/gauges.csv", FlowAccPath) -GaugesPath = "Hapi/Data/00inputs/Discharge/stations/" +Coello.read_gauge_table(Path + "/stations/gauges.csv", FlowAccPath) +GaugesPath = Path + "/stations/" Coello.read_discharge_gauges(GaugesPath, column='id', fmt="%Y-%m-%d") ``` ## 3-Run Object @@ -112,7 +112,7 @@ Run.run_distributed(Coello) ```python """ Outputs: - 1-statevariables: [numpy attribute] + 1-state_variables: 4D array (rows,cols,time,states) states are [sp,wc,sm,uz,lv] 2-qlz: [numpy attribute] 3D array of the lower zone discharge diff --git a/docs/examples/lumped-model-run.md b/docs/examples/lumped-model-run.md index 9092803e..2d03edc6 100644 --- a/docs/examples/lumped-model-run.md +++ b/docs/examples/lumped-model-run.md @@ -11,7 +11,7 @@ To run the HBV lumped model inside Hapi you need to prepare the meteorological i - First load the prepared lumped version of the HBV module inside Hapi, the triangular routing function and the wrapper function that runs the lumped model `RUN`. ```python -import hapi.rrm.hbv_bergestrom92 as HBVLumped +from hapi.rrm.hbv_bergestrom92 import HBVBergestrom92 as HBVLumped from hapi.run import Run from hapi.catchment import Catchment from hapi.routing import Routing @@ -79,6 +79,10 @@ all methods in `statista.descriptors` takes two numpy arrays of the same length ```python import statista.descriptors as metrics +# The observed record has to be read before it can be scored against; the page never +# loaded it, so `QGauges` was `None`. +Coello.read_discharge_gauges(Path + "Qout_c.csv", fmt="%Y-%m-%d") + Metrics = dict() Qobs = Coello.QGauges['q'] @@ -100,7 +104,7 @@ To plot the calculated and measured discharge import matplotlib gaugei = 0 plotstart = "2009-01-01" plotend = "2011-12-31" -Coello.plot_hydrograph(plotstart, plotend, gaugei, Title= "Lumped Model") +Coello.plot_hydrograph(plotstart, plotend, gaugei, title="Lumped Model") ``` ![lumped-model](../img/lumpedmodel.png) diff --git a/tests/rrm/catchment/test_read_parameters_validation.py b/tests/rrm/catchment/test_read_parameters_validation.py index 24423571..dc78a6c0 100644 --- a/tests/rrm/catchment/test_read_parameters_validation.py +++ b/tests/rrm/catchment/test_read_parameters_validation.py @@ -87,6 +87,62 @@ def test_unknown_resolution_is_rejected_at_construction(self): with pytest.raises(ValueError, match="temporal resolutions"): Catchment("coello", "2009-01-01", "2009-01-10", temporal_resolution="15min") + def test_hourly_resolution_scales_the_conversion_factor(self): + """Test that the hourly branch divides the daily conversion factor by 24. + + Test scenario: + The conversion factor turns depth per step into discharge, so it has to follow + the step length. The hourly value must be exactly a twenty-fourth of the daily + one, not a rounded constant. + """ + daily = Catchment("coello", "2009-01-01", "2009-01-10") + hourly = Catchment( + "coello", "2009-01-01", "2009-01-10", temporal_resolution="Hourly" + ) + + assert hourly.period.conversion_factor == pytest.approx( + daily.period.conversion_factor / 24 + ), ( + f"Expected {daily.period.conversion_factor / 24}, got " + f"{hourly.period.conversion_factor}" + ) + + +class TestSimulationPeriod: + """Tests for the span object every reader dates its inputs by.""" + + def test_a_span_that_runs_backwards_is_refused(self): + """Test that an end date before the start is named rather than silently empty. + + Test scenario: + A backwards span produces an empty `date_index`, which surfaces much later as a + zero-length driver mismatch naming neither date. + """ + with pytest.raises(ValueError, match="ends before it starts"): + SimulationPeriod.parse("2009-12-31", "2009-01-01") + + def test_the_calendar_is_built_once(self): + """Test that the derived calendar is cached rather than rebuilt on every read. + + Test scenario: + The class is frozen precisely so derived values cannot drift, which makes the + calendar safe to memoise -- yet `date_index`, and `days` and `__len__` through + it, rebuilt a `pd.date_range` on each access. `from_model` reads it once per + calibration trial and `SimulationResults._step_bounds` twice per call. + """ + period = SimulationPeriod.parse("2009-01-01", "2009-12-31") + + assert period.date_index is period.date_index, ( + "the calendar cannot change on a frozen period, so it should be built once" + ) + assert len(period) == len(period.date_index), ( + "the cached index must still be what the length reports" + ) + + +class TestConceptualModelInputs: + """Tests for the value objects the conceptual-model readers produce.""" + @pytest.mark.parametrize("initial_cond", [[0, 10, 10], [0] * 7, []]) def test_an_initial_condition_of_the_wrong_length_is_refused( self, initial_cond: list @@ -130,54 +186,6 @@ def test_bounds_of_different_lengths_are_refused(self): with pytest.raises(ValueError, match="same as LB"): ParameterBounds([0.0] * 12, [1.0] * 11) - def test_a_span_that_runs_backwards_is_refused(self): - """Test that an end date before the start is named rather than silently empty. - - Test scenario: - A backwards span produces an empty `date_index`, which surfaces much later as a - zero-length driver mismatch naming neither date. - """ - with pytest.raises(ValueError, match="ends before it starts"): - SimulationPeriod.parse("2009-12-31", "2009-01-01") - - def test_the_calendar_is_built_once(self): - """Test that the derived calendar is cached rather than rebuilt on every read. - - Test scenario: - The class is frozen precisely so derived values cannot drift, which makes the - calendar safe to memoise -- yet `date_index`, and `days` and `__len__` through - it, rebuilt a `pd.date_range` on each access. `from_model` reads it once per - calibration trial and `SimulationResults._step_bounds` twice per call. - """ - period = SimulationPeriod.parse("2009-01-01", "2009-12-31") - - assert period.date_index is period.date_index, ( - "the calendar cannot change on a frozen period, so it should be built once" - ) - assert len(period) == len(period.date_index), ( - "the cached index must still be what the length reports" - ) - - def test_hourly_resolution_scales_the_conversion_factor(self): - """Test that the hourly branch divides the daily conversion factor by 24. - - Test scenario: - The conversion factor turns depth per step into discharge, so it has to follow - the step length. The hourly value must be exactly a twenty-fourth of the daily - one, not a rounded constant. - """ - daily = Catchment("coello", "2009-01-01", "2009-01-10") - hourly = Catchment( - "coello", "2009-01-01", "2009-01-10", temporal_resolution="Hourly" - ) - - assert hourly.period.conversion_factor == pytest.approx( - daily.period.conversion_factor / 24 - ), ( - f"Expected {daily.period.conversion_factor / 24}, got " - f"{hourly.period.conversion_factor}" - ) - class TestReadParametersDistributed: """Tests for the distributed branch of `Catchment.read_parameters`.""" diff --git a/tests/sensitivity_analysis.py b/tests/sensitivity_analysis.py index a2aa833d..4113a7ea 100644 --- a/tests/sensitivity_analysis.py +++ b/tests/sensitivity_analysis.py @@ -35,10 +35,13 @@ parameters = pd.read_csv(Parameterpath, index_col=0, header=None) parameters.rename(columns={1: "value"}, inplace=True) # %% parameters boundaries -UB = pd.read_csv(Path + "/LB-1-Muskinguk.txt", index_col=0, header=None) +# Each bound read from its own file: `UB` came from `LB-...` and `LB` from `UB-...`, +# and both were handed straight to `SA(parameters, LB, UB, ...)`, so the sampler was +# given the range upside down. +UB = pd.read_csv(Path + "/UB-1-Muskinguk.txt", index_col=0, header=None) parnames = UB.index UB = UB[1].tolist() -LB = pd.read_csv(Path + "/UB-1-Muskinguk.txt", index_col=0, header=None) +LB = pd.read_csv(Path + "/LB-1-Muskinguk.txt", index_col=0, header=None) LB = LB[1].tolist() # The bounds moved onto `Calibration` with the is-a -> has-a change, and this script # only samples between them -- so it uses the two lists it just read. From 1c1efec47e4230cd1674252105f863d917f65ff1 Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Tue, 15 Sep 2026 20:14:29 +0200 Subject: [PATCH 50/54] docs(examples): say that the Jiboa example's dataset is not in the repository `examples/hydrological-model/jiboa/data/` is untracked -- 85 files, about 2 MB -- so the only exercise of `Run.run_distributed_with_lake` outside the unit tests fails on its first read for anyone who clones the branch. `tests/rrm/data/jiboa/` carries the lake record and its parameters and nothing else, which is not enough to drive it. The data stays out of the repository, by decision. The script's docstring now names exactly what it expects and says plainly that it is not shipped, so the failure is explained before it happens rather than surfacing as a missing-file traceback. --- .../Jiboa-distributed-model-muskingum-lake.py | 15 ++++++++++++--- 1 file changed, 12 insertions(+), 3 deletions(-) diff --git a/examples/hydrological-model/jiboa/Jiboa-distributed-model-muskingum-lake.py b/examples/hydrological-model/jiboa/Jiboa-distributed-model-muskingum-lake.py index 2cebde36..95042972 100644 --- a/examples/hydrological-model/jiboa/Jiboa-distributed-model-muskingum-lake.py +++ b/examples/hydrological-model/jiboa/Jiboa-distributed-model-muskingum-lake.py @@ -1,7 +1,16 @@ -"""This code is used to Run the distributed model for jiboa river in El Salvador where the catchment has an a ustream lake and a volcanic area. +"""Run the distributed model for the Jiboa river in El Salvador. -- you have to make the root directory to the examples folder to enable the code - from reading input files +The catchment has an upstream lake and a volcanic area, which makes it the only +exercise of `Run.run_distributed_with_lake` outside the unit tests. + +**The dataset this reads is not in the repository.** `examples/hydrological-model/jiboa/data/` +is untracked -- 85 files, about 2 MB: `lakedata.csv`, `Lakeparameters.txt`, `curve.txt`, +`initial-jiboa.txt`, `Initial-lake.txt`, and the `meteo-data/`, `gis-data/`, `parameters/` +and `gauges/` trees. A clone will not have it, and the script raises on the first read. +`tests/rrm/data/jiboa/` carries only the lake record and its parameters, which is not +enough to drive this. + +Run it from the repository root, so the relative paths below resolve. """ import datetime as dt From 00d052f0cee5b39209f1b52ec945300cc2091c8b Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Tue, 15 Sep 2026 20:23:30 +0200 Subject: [PATCH 51/54] test: cover the calibration guards, the one module this branch never measured Round 1 took seven modules to 100% line and branch. `calibration.py` was not among them and sat at 87% -- which is why round 2 found three of round 1's own fixes wrong there, and why one of them, the `ParameterBounds` width rule, broke every distributed calibration without a single test noticing. `tests/rrm/calibration/test_calibration_guards.py` covers what a calibration refuses before it starts, and the lumped objective loop that nothing exercised: * `read_parameters_bound` with a `snow` flag that is not a bool -- `0` and `1` hash equal to `False`/`True`, so an int gets the right answer by accident and a string fails much later; * `read_objective_function` with something that cannot be called, and the `args=None` default the entry points would otherwise try to unpack; * a missing search space, a missing objective, and each of the three inputs `_gauged_results` narrows; * an objective whose signature cannot be introspected, which the arity check has to decline to judge rather than reject; * `calibrate_lumped` with an incomplete `basic_inputs`, with no observed record, with a width the conceptual model cannot read, and with a wrongly wired objective -- each refused before the optimisation problem is declared; * and the trial loop itself: a valid trial runs the model, returns a finite score and both Muskingum stability constraints, while a trial that fails numerically is scored infeasible and the search continues. 87% -> 97%, with the other seven modules holding at 100%. What is left is `calibrate_maxbas`'s objective body, which needs a stubbed optimiser driving a distributed run, and three partial branches. 699 tests in the main task plus 17 in `plot`; mypy and ruff clean. --- .../calibration/test_calibration_guards.py | 465 ++++++++++++++++++ 1 file changed, 465 insertions(+) create mode 100644 tests/rrm/calibration/test_calibration_guards.py diff --git a/tests/rrm/calibration/test_calibration_guards.py b/tests/rrm/calibration/test_calibration_guards.py new file mode 100644 index 00000000..0f69b760 --- /dev/null +++ b/tests/rrm/calibration/test_calibration_guards.py @@ -0,0 +1,465 @@ +"""Tests for the refusals `Calibration` makes before and during a search. + +A calibration runs the model thousands of times, so the difference between a check that +fires at the call that got it wrong and one that fires on the first trial is the difference +between an error a caller can act on and a search that burns minutes before failing. These +are the guards that decide that, plus the lumped entry point's own objective loop -- the one +path where the optimiser's vector *is* the parameter set the conceptual model reads. +""" + +from __future__ import annotations + +import numpy as np +import pandas as pd +import pytest +import statista.descriptors as metrics + +from hapi.calibration import Calibration, ObjectiveFunctionArityError +from hapi.catchment import Catchment +from hapi.conceptual import ParameterBounds +from hapi.inputs import MeteoInputs +from hapi.results import RoutingKind, SimulationResults +from hapi.rrm.hbv_bergestrom92 import HBVBergestrom92 as HBVLumped + +CANNED_RESULT = (0.25, np.arange(12, dtype=float), {"time": 1.0}) + + +def _optimization_args() -> list: + """Build the three-element optimisation argument list the entry points unpack. + + Returns: + list: `[api_obj_args, pll_type, api_solve_args]`. + """ + return [ + dict(hms=2, hmcr=0.95, par=0.65, dbw=10, fileout=0, xinit=0, filename=""), + None, + dict(store_sol=False, display_opts=False, store_hst=False, hot_start=False), + ] + + +def _tiny_meteo() -> MeteoInputs: + """Build the smallest driver set the gauge accessor needs to get past its meteo guard. + + Returns: + MeteoInputs: Three `(2, 3, 4)` cubes. + """ + cube = np.zeros((2, 3, 4)) + return MeteoInputs(cube, cube.copy(), cube.copy()) + + +@pytest.fixture +def stub_engine(monkeypatch) -> dict: + """Replace the Harmony Search engine with a stub that drives the objective once. + + Args: + monkeypatch: Used to swap `HSapi` out of the calibration module. + + Returns: + dict: Records the objective's return value under `"scored"`. + """ + from hapi import calibration as calibration_module + + recorded: dict = {} + + class _StubEngine: + def __init__(self, *args, **kwargs): + pass + + def __call__(self, opt_prob, *args, **kwargs): + recorded["scored"] = opt_prob.obj_fun(np.full(12, 0.5)) + return CANNED_RESULT + + monkeypatch.setattr(calibration_module, "HSapi", _StubEngine) + return recorded + + +@pytest.fixture +def lumped( + coello_rrm_date: list, + lumped_meteo_data_path: str, + coello_AreaCoeff: float, + coello_InitialCond: list, + lumped_gauges_path: str, + coello_gauges_date_fmt: str, +) -> Calibration: + """A lumped calibration with drivers, a model, bounds and an observed record. + + Returns: + Calibration: Ready for `calibrate_lumped`. + """ + coello = Calibration(Catchment("rrm", coello_rrm_date[0], coello_rrm_date[1])) + coello.model.read_lumped_inputs(lumped_meteo_data_path) + coello.model.read_lumped_model(HBVLumped, coello_AreaCoeff, coello_InitialCond) + coello.model.read_discharge_gauges(lumped_gauges_path, fmt=coello_gauges_date_fmt) + coello.read_parameters_bound([0.0] * 12, [1.0] * 12, False) + coello.read_objective_function(metrics.rmse, []) + return coello + + +class TestReadParametersBound: + """The reader that settles the search space.""" + + @pytest.mark.parametrize("snow", [0, 1, "yes", None]) + def test_a_non_bool_snow_flag_is_refused( + self, coello_rrm_date: list, snow + ): + """Test that `snow` must be a bool, not something merely truthy. + + Args: + coello_rrm_date: Start and end dates for the model. + snow: A value that is not a bool. + + Test scenario: + `snow` selects the parameter count through `PARAMETER_COUNTS[(snow, maxbas)]`, + and `0`/`1` hash equal to `False`/`True` — so a caller passing an int gets the + right answer by accident and a caller passing a string gets a `KeyError` much + later. The type is checked where it enters. + """ + coello = Calibration(Catchment("rrm", coello_rrm_date[0], coello_rrm_date[1])) + + with pytest.raises(ValueError, match="True or False"): + coello.read_parameters_bound([0.0] * 12, [1.0] * 12, snow) + + +class TestReadObjectiveFunction: + """The reader that settles what a trial is scored with.""" + + def test_something_that_cannot_be_called_is_refused(self, coello_rrm_date: list): + """Test that a non-callable objective is named at the call that supplied it. + + Args: + coello_rrm_date: Start and end dates for the model. + + Test scenario: + Otherwise the failure is a `TypeError: 'str' object is not callable` from inside + the optimiser's first trial, several frames from the registration. + """ + coello = Calibration(Catchment("rrm", coello_rrm_date[0], coello_rrm_date[1])) + + with pytest.raises(TypeError, match="should be a function"): + coello.read_objective_function("rmse", []) + + def test_omitting_the_extra_arguments_leaves_an_empty_list( + self, coello_rrm_date: list + ): + """Test that `args=None` becomes an empty list rather than staying `None`. + + Args: + coello_rrm_date: Start and end dates for the model. + + Test scenario: + The entry points forward `*of_args`, which cannot unpack `None`. The default is + normalised here so every caller of `_objective()` gets something iterable. + """ + coello = Calibration(Catchment("rrm", coello_rrm_date[0], coello_rrm_date[1])) + + coello.read_objective_function(metrics.rmse, None) + + assert coello.OFArgs == [], f"expected [], got {coello.OFArgs!r}" + + +class TestTheGuardsBeforeTheSearch: + """What each entry point refuses before the optimisation problem is declared.""" + + def test_no_bounds_names_the_reader_that_supplies_them( + self, coello_rrm_date: list + ): + """Test that starting without a search space says which reader was skipped. + + Args: + coello_rrm_date: Start and end dates for the model. + + Test scenario: + Every entry point and the variable declaration read `self.bounds`, and all of + them used to index it straight — so a caller who forgot got `TypeError` on + `None`. + """ + coello = Calibration(Catchment("rrm", coello_rrm_date[0], coello_rrm_date[1])) + + with pytest.raises(ValueError, match="read_parameters_bound"): + coello._search_space() + + def test_no_objective_names_the_reader_that_supplies_it( + self, coello_rrm_date: list + ): + """Test that starting without an objective says which reader was skipped. + + Args: + coello_rrm_date: Start and end dates for the model. + + Test scenario: + The same shape as the bounds guard: the objective is read once per trial, so + reporting it up front is what keeps a caller from waiting for the first one. + """ + coello = Calibration(Catchment("rrm", coello_rrm_date[0], coello_rrm_date[1])) + + with pytest.raises(ValueError, match="read_objective_function"): + coello._objective() + + def test_an_objective_with_no_readable_signature_is_left_alone( + self, coello_rrm_date: list + ): + """Test that a builtin without an introspectable signature skips the arity check. + + Args: + coello_rrm_date: Start and end dates for the model. + + Test scenario: + `inspect.signature` raises for callables that decline to describe themselves — + some C functions, and anything wrapping them. Refusing those would reject a + legitimate objective for being unreadable, so the check declines to judge rather + than guessing; the runtime handler still scores a bad call as infeasible. + """ + + class _Unreadable: + """A callable that refuses to describe its own signature.""" + + @property + def __signature__(self): + """Raise the way an un-introspectable callable does.""" + raise ValueError("no signature for this one") + + def __call__(self, *args): + """Accept anything.""" + return 0.0 + + coello = Calibration(Catchment("rrm", coello_rrm_date[0], coello_rrm_date[1])) + coello.read_objective_function(_Unreadable(), []) + + coello._check_objective_arity(99) + + @pytest.mark.parametrize( + "missing, basic_inputs", + [ + ("RoutingFn", {"Route": 1}), + ("Route", {"RoutingFn": None}), + ("Route", {}), + ], + ) + def test_incomplete_basic_inputs_name_the_missing_key( + self, lumped: Calibration, missing: str, basic_inputs: dict + ): + """Test that `calibrate_lumped` names the key its bundle is missing. + + Args: + lumped: A lumped calibration ready to run. + missing: A key the bundle omits. + basic_inputs: The incomplete bundle. + + Test scenario: + `basic_inputs` is a loose dict, so a typo is a missing key rather than a missing + argument, and the failure would otherwise be a `KeyError` inside the objective + once per trial. + """ + with pytest.raises(ValueError, match=missing): + lumped.calibrate_lumped(basic_inputs, _optimization_args()) + + def test_no_observed_record_names_the_reader_that_supplies_it( + self, + coello_rrm_date: list, + lumped_meteo_data_path: str, + coello_AreaCoeff: float, + coello_InitialCond: list, + stub_engine: dict, + ): + """Test that a lumped calibration without gauges says what it cannot score against. + + Args: + coello_rrm_date: Start and end dates for the model. + lumped_meteo_data_path: Driver record. + coello_AreaCoeff: Catchment area. + coello_InitialCond: Initial state. + stub_engine: Drives the objective once. + + Test scenario: + The objective is handed `QGauges` as its first argument. Unread, that is `None`, + and the metric fails on it inside a trial that is then scored infeasible — so + the whole search reports nothing but `nan`. + """ + coello = Calibration(Catchment("rrm", coello_rrm_date[0], coello_rrm_date[1])) + coello.model.read_lumped_inputs(lumped_meteo_data_path) + coello.model.read_lumped_model(HBVLumped, coello_AreaCoeff, coello_InitialCond) + coello.read_parameters_bound([0.0] * 12, [1.0] * 12, False) + coello.read_objective_function(metrics.rmse, []) + + with pytest.raises(ValueError, match="read_discharge_gauges"): + coello.calibrate_lumped( + {"Route": 0, "RoutingFn": None}, _optimization_args() + ) + + +class TestGaugedResults: + """The accessor every gauge-reading path narrows through.""" + + @pytest.mark.parametrize( + "clear, expected", + [ + ("results", "no trial has completed"), + ("meteo", "no drivers"), + ("GaugesTable", "read_gauge_table"), + ], + ) + def test_each_missing_input_is_named( + self, lumped: Calibration, clear: str, expected: str + ): + """Test that each of the three inputs is reported by the reader that supplies it. + + Args: + lumped: A lumped calibration ready to run. + clear: The attribute to clear on the model. + expected: Substring the error must carry. + + Test scenario: + All three are indexed straight afterwards, so an unset one used to fail on + `None` inside the extraction loop -- naming an attribute of an array rather than + the step nobody took. + """ + lumped.model.results = SimulationResults( + RoutingKind.MUSKINGUM, + np.zeros((2, 3, 4), dtype="float32"), + np.zeros((2, 3, 4), dtype="float32"), + None, + q_total=np.zeros((2, 3, 4), dtype="float32"), + ) + lumped.model.meteo = _tiny_meteo() + lumped.model.GaugesTable = pd.DataFrame({"cell_row": [0], "cell_col": [0]}) + setattr(lumped.model, clear, None) + + with pytest.raises(ValueError, match=expected): + lumped._gauged_results() + + def test_results_with_no_routed_discharge_are_refused(self, lumped: Calibration): + """Test that routed-looking results carrying no `q_total` say so. + + Args: + lumped: A lumped calibration ready to run. + + Test scenario: + The routing kind and the arrays are set separately, so a results object can + claim a routing it does not have the fields for -- which is a different failure + from "nothing routed these", and gets its own message. + """ + lumped.model.results = SimulationResults( + RoutingKind.MUSKINGUM, + np.zeros((2, 3, 4), dtype="float32"), + np.zeros((2, 3, 4), dtype="float32"), + None, + ) + lumped.model.meteo = _tiny_meteo() + lumped.model.GaugesTable = pd.DataFrame({"cell_row": [0], "cell_col": [0]}) + + with pytest.raises(ValueError, match="no routed discharge"): + lumped.extract_discharge() + + +class TestCalibrateLumped: + """The one entry point where the optimiser's vector is the parameter set itself.""" + + def test_a_trial_runs_the_model_and_scores_it( + self, lumped: Calibration, stub_engine: dict + ): + """Test that the objective loop runs the model and returns a finite score. + + Args: + lumped: A lumped calibration ready to run. + stub_engine: Drives the objective once and records what it returned. + + Test scenario: + This is the body nothing exercised: the trial installs its parameters, runs the + lumped model, scores `Qsim` against the observed record, and builds the two + Muskingum stability constraints. A stubbed optimiser is what makes it a unit + test rather than a search. + """ + result = lumped.calibrate_lumped( + {"Route": 0, "RoutingFn": None}, _optimization_args() + ) + + assert result is CANNED_RESULT, "the optimiser's result comes back untouched" + error, constraints, fail = stub_engine["scored"] + assert fail == 0, f"a valid trial must not be scored infeasible, got {fail}" + assert np.isfinite(error), f"the score must be a real number, got {error}" + assert len(constraints) == 2, ( + f"the two Muskingum stability constraints must be returned, got {constraints}" + ) + + def test_a_failing_trial_is_scored_infeasible_rather_than_ending_the_search( + self, lumped: Calibration, stub_engine: dict + ): + """Test that a trial that blows up numerically does not stop the calibration. + + Args: + lumped: A lumped calibration ready to run. + stub_engine: Drives the objective once and records what it returned. + + Test scenario: + One bad candidate out of thousands is normal, and the handler exists so the + search survives it. The arity escape added this round had to not break that. + """ + + def explodes(observed, simulated): + """Fail on the values, with the arity the call site passes.""" + raise ZeroDivisionError("no discharge at this gauge") + + lumped.read_objective_function(explodes, []) + + lumped.calibrate_lumped({"Route": 0, "RoutingFn": None}, _optimization_args()) + + error, constraints, fail = stub_engine["scored"] + assert fail == 1, "a failing trial is reported as infeasible" + assert np.isnan(error), f"an infeasible trial scores nan, got {error}" + assert constraints == [], f"no constraints survive a failed trial, got {constraints}" + + def test_a_wrongly_wired_objective_stops_before_the_optimiser_is_built( + self, lumped: Calibration, stub_engine: dict + ): + """Test that the lumped path checks arity too, before declaring the problem. + + Args: + lumped: A lumped calibration ready to run. + stub_engine: Records whether the engine was reached. + + Test scenario: + The lumped call shape is `objective(observed, Qsim, *of_args)`. An objective that + cannot accept that many is a wiring error, not a bad candidate, and every trial + would fail identically. + """ + + def needs_five(a, b, c, d, e): + """Take more arguments than the entry point passes.""" + return 0.0 + + lumped.read_objective_function(needs_five, []) + + with pytest.raises(ObjectiveFunctionArityError, match="needs more inputs"): + lumped.calibrate_lumped( + {"Route": 0, "RoutingFn": None}, _optimization_args() + ) + + assert "scored" not in stub_engine, ( + "the optimiser must not be built when the objective cannot be called" + ) + + def test_a_search_width_the_model_cannot_read_is_refused( + self, lumped: Calibration, stub_engine: dict + ): + """Test that the lumped width rule fires before the problem is declared. + + Args: + lumped: A lumped calibration ready to run. + stub_engine: Records whether the engine was reached. + + Test scenario: + A lumped calibration is the one case where the optimiser's vector *is* the + parameter set, so its width has to match `PARAMETER_COUNTS`. The rule cannot + live on `ParameterBounds`, which also bounds distributed searches of 980 values. + """ + lumped.bounds = ParameterBounds(np.zeros(9), np.ones(9)) + + with pytest.raises(ValueError, match="takes 12 parameters"): + lumped.calibrate_lumped( + {"Route": 0, "RoutingFn": None}, _optimization_args() + ) + + assert "scored" not in stub_engine, ( + "the optimiser must not be built for a width the model cannot read" + ) From 18051cc367ae87265d7a91689469f6b0b350c074 Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Tue, 15 Sep 2026 20:26:26 +0200 Subject: [PATCH 52/54] docs: state the three contracts this round's fixes changed `read_objective_function`'s `args` were documented as "forwarded to the objective" and two of the three entry points silently dropped them. They are forwarded now, so the docstring says what the caller can rely on -- appended after the arguments the entry point supplies, counted by the arity check, so registering arguments the objective cannot accept is reported before the search starts rather than scored as `nan`. Two runnable examples: the arguments are kept as given, and a non-callable is refused where it is registered. `calibrate_lumped` gained the parameter-count rule this round, and its `Raises:` now says so and why it lives there rather than on `ParameterBounds` -- a distributed calibration searches `SpatialVarFun.ParametersNO` values, so the widths coincide only on this path. It also documents `ObjectiveFunctionArityError`, which it can raise before the optimiser is built. `route_maxbas_by_path_length` refuses a constant flow-path-length raster, and `Catchment.extract_discharge` refuses MAXBAS results carrying no `qout`. Both `Raises:` sections name the new condition and, for the second, where `qout` is supposed to come from -- the `Wrapper` entry points, not the routers. 47 doctests pass, up from 46; `mkdocs build --strict` stays clean. --- src/hapi/calibration.py | 44 +++++++++++++++++++++++++++++++++++++---- src/hapi/catchment.py | 9 ++++++--- src/hapi/rrm/distrrm.py | 4 +++- 3 files changed, 49 insertions(+), 8 deletions(-) diff --git a/src/hapi/calibration.py b/src/hapi/calibration.py index 9ca6184c..91408877 100644 --- a/src/hapi/calibration.py +++ b/src/hapi/calibration.py @@ -387,11 +387,39 @@ def read_objective_function( Args: objective_function (callable): A callable function to calculate any kind of metric to be used in the calibration. - args: Any positional or keyword arguments to pass to the - objective function. If None, defaults to an empty list. + args: Extra positional arguments appended to every call of the objective, + after the ones the entry point supplies itself. All three entry points + forward them, and the arity check counts them, so registering arguments the + objective cannot accept is reported before the search starts. `None` + becomes an empty list. Raises: TypeError: If objective_function is not callable. + + Examples: + - The arguments are kept as given, and `None` becomes an empty list: + ```python + >>> import statista.descriptors as metrics + >>> from hapi.calibration import Calibration + >>> from hapi.catchment import Catchment + >>> coello = Calibration(Catchment("coello", "2009-01-01", "2009-01-10")) + >>> coello.read_objective_function(metrics.rmse, [0.5, "outlet"]) + Objective function is read successfully + >>> coello.OFArgs + [0.5, 'outlet'] + + ``` + - Something that cannot be called is refused where it is registered: + ```python + >>> from hapi.calibration import Calibration + >>> from hapi.catchment import Catchment + >>> coello = Calibration(Catchment("coello", "2009-01-01", "2009-01-10")) + >>> coello.read_objective_function("rmse", []) + Traceback (most recent call last): + ... + TypeError: The Objective function should be a function, got str + + ``` """ # check objective_function if not callable(objective_function): @@ -812,8 +840,16 @@ def calibrate_lumped( Raises: ValueError: If `basic_inputs` is missing required keys - `"Route"` or `"RoutingFn"`, or if `"InitialValues"` is - given and does not hold one value per parameter. + `"Route"` or `"RoutingFn"`, if `"InitialValues"` is + given and does not hold one value per parameter, or if the search space is + not the width the conceptual model reads. That last rule is checked here + rather than on :class:`~hapi.conceptual.ParameterBounds` because this is the + one entry point where the optimiser's vector *is* the parameter set: a + distributed calibration searches `SpatialVarFun.ParametersNO` values and + maps them onto the grid. + ObjectiveFunctionArityError: The objective cannot be called with the arguments + this entry point passes -- `observed`, `Qsim`, and whatever + :meth:`read_objective_function` registered. TypeError: If either bundle of optimization arguments is not a dict. """ diff --git a/src/hapi/catchment.py b/src/hapi/catchment.py index abb20ec8..6fdc6272 100644 --- a/src/hapi/catchment.py +++ b/src/hapi/catchment.py @@ -1046,9 +1046,12 @@ def extract_discharge(self, calculate_metrics=True, factor=None): per-gauge (Muskingum) path. Default is None. Raises: - ValueError: The gauge table has not been read, the model has not been run, or - the results it produced have not been routed -- there is no hydrograph to - extract from a set of arrays no routing step has filled. + ValueError: The gauge table has not been read, the model has not been run, the + results it produced have not been routed, or -- on a MAXBAS run -- they + carry no `qout`. The routers record the routing they applied but do not sum + the domain, so results routed by calling + :class:`~hapi.rrm.distrrm.DistributedRRM` directly reach here labelled + MAXBAS with no outlet series; the `Wrapper` entry points are what fill it. """ if self.GaugesTable is None: raise ValueError("please read the gauges' table first.") diff --git a/src/hapi/rrm/distrrm.py b/src/hapi/rrm/distrrm.py index da3e5385..62d536eb 100644 --- a/src/hapi/rrm/distrrm.py +++ b/src/hapi/rrm/distrrm.py @@ -268,7 +268,9 @@ def route_maxbas_by_path_length( results: The results to route. Mutated in place. Raises: - ValueError: The run carries no flow-path-length raster. + ValueError: The run carries no flow-path-length raster, or the raster is + constant -- MAXBAS is scaled along the spread of distances to the outlet, + so a raster with no spread has nothing to scale along. """ if run.flow_path_length is None: raise ValueError( From f401c0a0cde860f1dc27172f9d052c5eab014b74 Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Tue, 15 Sep 2026 20:27:32 +0200 Subject: [PATCH 53/54] style(tests): apply ruff-format to the calibration guard tests --- tests/rrm/calibration/test_calibration_guards.py | 12 +++++------- 1 file changed, 5 insertions(+), 7 deletions(-) diff --git a/tests/rrm/calibration/test_calibration_guards.py b/tests/rrm/calibration/test_calibration_guards.py index 0f69b760..271904ac 100644 --- a/tests/rrm/calibration/test_calibration_guards.py +++ b/tests/rrm/calibration/test_calibration_guards.py @@ -100,9 +100,7 @@ class TestReadParametersBound: """The reader that settles the search space.""" @pytest.mark.parametrize("snow", [0, 1, "yes", None]) - def test_a_non_bool_snow_flag_is_refused( - self, coello_rrm_date: list, snow - ): + def test_a_non_bool_snow_flag_is_refused(self, coello_rrm_date: list, snow): """Test that `snow` must be a bool, not something merely truthy. Args: @@ -161,9 +159,7 @@ def test_omitting_the_extra_arguments_leaves_an_empty_list( class TestTheGuardsBeforeTheSearch: """What each entry point refuses before the optimisation problem is declared.""" - def test_no_bounds_names_the_reader_that_supplies_them( - self, coello_rrm_date: list - ): + def test_no_bounds_names_the_reader_that_supplies_them(self, coello_rrm_date: list): """Test that starting without a search space says which reader was skipped. Args: @@ -407,7 +403,9 @@ def explodes(observed, simulated): error, constraints, fail = stub_engine["scored"] assert fail == 1, "a failing trial is reported as infeasible" assert np.isnan(error), f"an infeasible trial scores nan, got {error}" - assert constraints == [], f"no constraints survive a failed trial, got {constraints}" + assert constraints == [], ( + f"no constraints survive a failed trial, got {constraints}" + ) def test_a_wrongly_wired_objective_stops_before_the_optimiser_is_built( self, lumped: Calibration, stub_engine: dict From 028efd6dc21a41340ab1b6c0749b320676279bb1 Mon Sep 17 00:00:00 2001 From: Mostafa Farrag Date: Tue, 15 Sep 2026 20:36:57 +0200 Subject: [PATCH 54/54] refactor(catchment): score a gauge in one place, and name the repeated log line The SonarCloud sweep on PR 217. The quality gate passes on every condition, and two of its six CRITICAL findings are real and caused by this branch. `Catchment.extract_discharge` tipped over the cognitive-complexity limit (18 against 15) when this round added the UNROUTED refusal and the `qout` requirement. The cause was already there: the seven-metric block was written out twice, once per routing branch, which is also how the two would drift apart. `GAUGE_METRICS` names the seven once and `_score_gauge` fills a gauge's column from it, so both branches now say `_score_gauge(...)` and the method is back under the limit. `run.py` logged the literal "Model Run has finished" from three entry points; it is `RUN_FINISHED` now. Seven `pytest.raises` blocks in the tests this branch added wrapped more than one call that could throw -- `_optimization_args()`, `LumpedRun.from_model(...)`, `dt.datetime(...)` -- so a failure in the setup would have read as the refusal under test. The setup is hoisted out; only the call being tested is inside the block. The other four CRITICAL findings are false positives, reported rather than marked: `python:S5655` on `Run.run_flood`, `Run.run_distributed_with_lake` and `Calibration.run_calibration` claims the arguments are the wrong type. Those parameters are annotated with `typing.Protocol` classes (`CatchmentLike`, `SpatialDistribution`) that `Catchment` and `Parameters` satisfy structurally, which is the whole point of `hapi.protocols` -- and mypy checks all 28 modules clean. --- src/hapi/catchment.py | 77 +++++++++---------- src/hapi/run.py | 9 ++- .../test_calibration_distributed.py | 4 +- .../calibration/test_calibration_guards.py | 22 +++--- .../catchment/test_maxbas_routing_variants.py | 6 +- tests/rrm/catchment/test_results.py | 10 ++- 6 files changed, 67 insertions(+), 61 deletions(-) diff --git a/src/hapi/catchment.py b/src/hapi/catchment.py index 6fdc6272..6d76704f 100644 --- a/src/hapi/catchment.py +++ b/src/hapi/catchment.py @@ -21,7 +21,7 @@ from collections.abc import Iterator from contextlib import contextmanager from pathlib import Path -from typing import TYPE_CHECKING, Self +from typing import TYPE_CHECKING, Any, Self import matplotlib.dates as dates import matplotlib.pyplot as plt @@ -209,6 +209,33 @@ def _name_the_path(path) -> Iterator[None]: raise FileNotFoundError(f"{exc} (path: {path})") from exc +#: The seven performance metrics `extract_discharge` scores a gauge with, in the order the +#: `metrics` frame indexes them. Named once because both routing branches compute the same +#: seven, and writing them out twice is how the two drift apart. +GAUGE_METRICS: dict[str, Any] = { + "RMSE": metrics.rmse, + "NSE": metrics.nse, + "NSEhf": metrics.nse_hf, + "KGE": metrics.kge, + "WB": metrics.wb, + "Pearson-CC": metrics.pearson_corr_coeff, + "R2": metrics.r2, +} + + +def _score_gauge(frame: pd.DataFrame, gauge_id: Any, q_obs, q_sim) -> None: + """Fill one gauge's column of a metrics frame. + + Args: + frame: The metrics frame, indexed by :data:`GAUGE_METRICS`' keys. + gauge_id: The column to write. + q_obs: Observed discharge at that gauge. + q_sim: Simulated discharge at that gauge. + """ + for name, metric in GAUGE_METRICS.items(): + frame.loc[name, gauge_id] = round(metric(q_obs, q_sim), 3) + + class Catchment: """Catchment for reading meteorological/spatial inputs and running the model. @@ -1071,8 +1098,9 @@ def extract_discharge(self, calculate_metrics=True, factor=None): index=self.period.date_index, columns=self.QGauges.columns ) if calculate_metrics: - index = ["RMSE", "NSE", "NSEhf", "KGE", "WB", "Pearson-CC", "R2"] - self.metrics = pd.DataFrame(index=index, columns=self.QGauges.columns) + self.metrics = pd.DataFrame( + index=list(GAUGE_METRICS), columns=self.QGauges.columns + ) # sum the lower zone and the upper zone discharge outlet_x = self.flow_network.outlet[0][0] outlet_y = self.flow_network.outlet[1][0] @@ -1104,27 +1132,8 @@ def extract_discharge(self, calculate_metrics=True, factor=None): self.Qsim.loc[:, gauge_id] = q_sim if calculate_metrics: - q_obs = self.QGauges.loc[:, gauge_id] - self.metrics.loc["RMSE", gauge_id] = round( - metrics.rmse(q_obs, q_sim), 3 - ) - self.metrics.loc["NSE", gauge_id] = round( - metrics.nse(q_obs, q_sim), 3 - ) - self.metrics.loc["NSEhf", gauge_id] = round( - metrics.nse_hf(q_obs, q_sim), 3 - ) - self.metrics.loc["KGE", gauge_id] = round( - metrics.kge(q_obs, q_sim), 3 - ) - self.metrics.loc["WB", gauge_id] = round( - metrics.wb(q_obs, q_sim), 3 - ) - self.metrics.loc["Pearson-CC", gauge_id] = round( - metrics.pearson_corr_coeff(q_obs, q_sim), 3 - ) - self.metrics.loc["R2", gauge_id] = round( - metrics.r2(q_obs, q_sim), 3 + _score_gauge( + self.metrics, gauge_id, self.QGauges.loc[:, gauge_id], q_sim ) else: # MAXBAS: a cell of `q_total` is a contribution, so the hydrograph is the @@ -1142,24 +1151,10 @@ def extract_discharge(self, calculate_metrics=True, factor=None): self.Qsim.loc[:, gauge_id] = q_sim if calculate_metrics: - index = ["RMSE", "NSE", "NSEhf", "KGE", "WB", "Pearson-CC", "R2"] - self.metrics = pd.DataFrame(index=index) - - # if CalculateMetrics: - q_obs = self.QGauges.loc[:, gauge_id] - self.metrics.loc["RMSE", gauge_id] = round( - metrics.rmse(q_obs, q_sim), 3 - ) - self.metrics.loc["NSE", gauge_id] = round(metrics.nse(q_obs, q_sim), 3) - self.metrics.loc["NSEhf", gauge_id] = round( - metrics.nse_hf(q_obs, q_sim), 3 - ) - self.metrics.loc["KGE", gauge_id] = round(metrics.kge(q_obs, q_sim), 3) - self.metrics.loc["WB", gauge_id] = round(metrics.wb(q_obs, q_sim), 3) - self.metrics.loc["Pearson-CC", gauge_id] = round( - metrics.pearson_corr_coeff(q_obs, q_sim), 3 + self.metrics = pd.DataFrame(index=list(GAUGE_METRICS)) + _score_gauge( + self.metrics, gauge_id, self.QGauges.loc[:, gauge_id], q_sim ) - self.metrics.loc["R2", gauge_id] = round(metrics.r2(q_obs, q_sim), 3) def plot_hydrograph( self, diff --git a/src/hapi/run.py b/src/hapi/run.py index 99a0fa9b..4caa522d 100644 --- a/src/hapi/run.py +++ b/src/hapi/run.py @@ -33,6 +33,9 @@ if TYPE_CHECKING: from hapi.catchment import Lake as LakeType +#: Logged by each distributed entry point when it returns. +RUN_FINISHED = "Model Run has finished" + def _check_lake_meteo(run: DistributedRun, lake: LakeType) -> None: """Check the lake's record lines up with the distributed drivers. @@ -182,7 +185,7 @@ def run_distributed(model: CatchmentLike) -> SimulationResults: results = Wrapper.run_muskingum(run) model.results = results - logger.info("Model Run has finished") + logger.info(RUN_FINISHED) return results @staticmethod @@ -271,7 +274,7 @@ def run_distributed_with_lake( results = Wrapper.run_muskingum_with_lake(run, lake) model.results = results - logger.info("Model Run has finished") + logger.info(RUN_FINISHED) return results @staticmethod @@ -307,7 +310,7 @@ def run_maxbas(model: CatchmentLike) -> SimulationResults: results = Wrapper.run_maxbas(run) model.results = results - logger.info("Model Run has finished") + logger.info(RUN_FINISHED) return results @staticmethod diff --git a/tests/rrm/calibration/test_calibration_distributed.py b/tests/rrm/calibration/test_calibration_distributed.py index 9efabeba..2745acf1 100644 --- a/tests/rrm/calibration/test_calibration_distributed.py +++ b/tests/rrm/calibration/test_calibration_distributed.py @@ -254,8 +254,10 @@ def needs_four_arguments(observed, simulated, gauges, extra): coello.read_objective_function(needs_four_arguments, []) + args = _optimization_args() + with pytest.raises(ObjectiveFunctionArityError, match="needs more inputs"): - coello.run_calibration(spatial_var_stub, _optimization_args()) + coello.run_calibration(spatial_var_stub, args) def test_a_type_error_from_inside_a_correct_objective_is_not_a_wiring_error( self, gauged_calibration: Calibration, stub_optimizer: dict, spatial_var_stub diff --git a/tests/rrm/calibration/test_calibration_guards.py b/tests/rrm/calibration/test_calibration_guards.py index 271904ac..04022080 100644 --- a/tests/rrm/calibration/test_calibration_guards.py +++ b/tests/rrm/calibration/test_calibration_guards.py @@ -247,8 +247,10 @@ def test_incomplete_basic_inputs_name_the_missing_key( argument, and the failure would otherwise be a `KeyError` inside the objective once per trial. """ + args = _optimization_args() + with pytest.raises(ValueError, match=missing): - lumped.calibrate_lumped(basic_inputs, _optimization_args()) + lumped.calibrate_lumped(basic_inputs, args) def test_no_observed_record_names_the_reader_that_supplies_it( self, @@ -278,10 +280,10 @@ def test_no_observed_record_names_the_reader_that_supplies_it( coello.read_parameters_bound([0.0] * 12, [1.0] * 12, False) coello.read_objective_function(metrics.rmse, []) + args = _optimization_args() + with pytest.raises(ValueError, match="read_discharge_gauges"): - coello.calibrate_lumped( - {"Route": 0, "RoutingFn": None}, _optimization_args() - ) + coello.calibrate_lumped({"Route": 0, "RoutingFn": None}, args) class TestGaugedResults: @@ -428,10 +430,10 @@ def needs_five(a, b, c, d, e): lumped.read_objective_function(needs_five, []) + args = _optimization_args() + with pytest.raises(ObjectiveFunctionArityError, match="needs more inputs"): - lumped.calibrate_lumped( - {"Route": 0, "RoutingFn": None}, _optimization_args() - ) + lumped.calibrate_lumped({"Route": 0, "RoutingFn": None}, args) assert "scored" not in stub_engine, ( "the optimiser must not be built when the objective cannot be called" @@ -453,10 +455,10 @@ def test_a_search_width_the_model_cannot_read_is_refused( """ lumped.bounds = ParameterBounds(np.zeros(9), np.ones(9)) + args = _optimization_args() + with pytest.raises(ValueError, match="takes 12 parameters"): - lumped.calibrate_lumped( - {"Route": 0, "RoutingFn": None}, _optimization_args() - ) + lumped.calibrate_lumped({"Route": 0, "RoutingFn": None}, args) assert "scored" not in stub_engine, ( "the optimiser must not be built for a width the model cannot read" diff --git a/tests/rrm/catchment/test_maxbas_routing_variants.py b/tests/rrm/catchment/test_maxbas_routing_variants.py index a2ba49ea..a6c19826 100644 --- a/tests/rrm/catchment/test_maxbas_routing_variants.py +++ b/tests/rrm/catchment/test_maxbas_routing_variants.py @@ -294,10 +294,10 @@ def test_lumped_routing_refuses_something_that_cannot_be_called( coello_InitialCond, ) + run = LumpedRun.from_model(model) + with pytest.raises(TypeError, match="callable"): - Wrapper.run_lumped( - LumpedRun.from_model(model), Routing=1, RoutingFn=routing_fn - ) + Wrapper.run_lumped(run, Routing=1, RoutingFn=routing_fn) class TestLumpedRouting: diff --git a/tests/rrm/catchment/test_results.py b/tests/rrm/catchment/test_results.py index 654a9313..1546a886 100644 --- a/tests/rrm/catchment/test_results.py +++ b/tests/rrm/catchment/test_results.py @@ -499,12 +499,16 @@ def test_datetime_arguments_are_accepted_as_they_are( `fmt`; a datetime must skip that, since `strptime` on one raises `TypeError`. Reaching the *result*-option error proves both bounds resolved. """ + start = dt.datetime(2009, 1, 1) + end = dt.datetime(2009, 1, 5) + destination = str(tmp_path) + with pytest.raises(ValueError, match="between 1 and 8"): unrouted.save( - path=str(tmp_path), + path=destination, result=99, - start=dt.datetime(2009, 1, 1), - end=dt.datetime(2009, 1, 5), + start=start, + end=end, flow_acc_path="unused", )