diff --git a/src/ispypsa/templater/create_template.py b/src/ispypsa/templater/create_template.py index aa8f991b..cdc45900 100644 --- a/src/ispypsa/templater/create_template.py +++ b/src/ispypsa/templater/create_template.py @@ -17,6 +17,7 @@ ) from ispypsa.templater.existing_planned import ( _template_generators_existing_planned, + _template_storage_existing_planned, ) from ispypsa.templater.filter_template import _filter_template from ispypsa.templater.flow_paths import ( @@ -101,6 +102,7 @@ "costs_connection", "generators_existing_planned", "generators_new_entrant", + "storage_existing_planned", "storage_new_entrant", "custom_constraints", "custom_constraints_lhs", @@ -257,6 +259,9 @@ def create_ispypsa_inputs_template( template["generators_existing_planned"] = _template_generators_existing_planned( iasr_tables, regional_granularity, sub_regional_geography ) + template["storage_existing_planned"] = _template_storage_existing_planned( + iasr_tables, regional_granularity, sub_regional_geography + ) if regional_granularity == "sub_regions": template.update( diff --git a/src/ispypsa/templater/existing_planned.py b/src/ispypsa/templater/existing_planned.py index 7a6e23a8..6b933a05 100644 --- a/src/ispypsa/templater/existing_planned.py +++ b/src/ispypsa/templater/existing_planned.py @@ -5,35 +5,40 @@ Both target tables — see schemas/generators_existing_planned.yaml and schemas/storage_existing_planned.yaml — are built from the single IASR existing_committed_anticipated_additional_generator_summary table, which already lists -one row per real generating/storage unit (DUID-level). TODO: finish templating storage. +one row per real generating/storage unit (DUID-level). existing_committed_anticipated_additional_generator_summary: IASR ID / DLT names Power Station Technology Type REZ ID Sub-region Fuel type Fuel cost mapping BW01 Bayswater Steam Sub Critical NA CNSW Coal Bayswater - Q8 Battery - 2h Q8 Battery Battery Storage... Q8 SQ - - + DALNTH1 Dalrymple BESS Battery Storage... S4 CSA - - generators_existing_planned (partial): - name power_station technology geo_id fuel_type fuel_price_mapping capacity + name power_station technology geo_id fuel_type fuel_price_mapping capacity BW01 Bayswater Steam Sub Critical CNSW Coal Bayswater 660.0 - storage_existing_planned (partial, identity only): - name power_station technology - Q8 Battery - 2h Q8 Battery Battery Storage (2hrs storage) - -Building generators_existing_planned: - 1. Splits the summary's rows into generators and storage — see - _is_existing_planned_storage_row. - 2. Renames the carried-over spine/identity columns to their schema names, derives - geo_id (REZ ID with Sub-region fallback — see helpers._set_geo_id) and relabels - it to ``regional_granularity`` (REZ-located rows stay untouched at every - granularity). - 3. Merges in unit-level properties (mappings.py). Each generator's - ``name`` is resolved against a source table's own IASR ID column — exact - matches first, small typos fuzzy-corrected. Every existing/planned unit is - expected to resolve to a real row in each of these tables; an unresolved name raises. - 4. Merges in minimum_load: coal's Typical Lowest Band, then gas overlaid - (see _merge_minimum_load) — the only two technologies with published minimum - stable levels, so every other row is left NaN (expected). + storage_existing_planned (partial): + name power_station technology geo_id fuel_type capacity storage_capacity + DALNTH1 Dalrymple BESS Battery Storage... S4 Battery 30 9 + +Building {generators/storage}_existing_planned: + 1. Splits the summary's rows into generators and storage — see + _is_existing_planned_storage_row. + 2. Renames the carried-over spine/identity columns to their schema names, derives + geo_id (REZ ID with Sub-region fallback — see helpers._set_geo_id) and relabels + it to ``regional_granularity`` (REZ-located rows stay untouched at every + granularity). + 3. Merges in unit-level properties (mappings.py). Each unit's ``name`` is resolved + against a source table's own IASR ID column — exact matches first, small + typos fuzzy-corrected. Every existing/planned unit is expected to resolve to + a real row in each of these tables; an unresolved name raises. + 4. For storage units - then merges in category-level properties: ``power_station`` or + ``technology`` summary columns are resolved against equivalent columns in + the property table, using those categories to map values onto. + 5. For generators - merges in minimum_load: coal's Typical Lowest Band, then gas overlaid + (see _merge_minimum_load) — the only two technologies with published minimum + stable levels, so every other row is left NaN (expected). + 6. Returns the filled out summary tables with only required columns present + (see ``_GENERATOR_COLUMNS`` and ``_STORAGE_COLUMNS`` below). """ import logging @@ -41,17 +46,25 @@ import pandas as pd from ispypsa.templater.helpers import ( - _apply_known_value_replacement, + _apply_iasr_table_replacements, _assert_table_valid, + _derive_phes_symmetric_efficiency, _fuzzy_match_names, _get_property_value_map, _group_properties_by_source, + _is_battery_row, _is_storage_row, _map_geo_id_to_granularity, + _merge_category_keyed_properties, _required_property_columns, _set_geo_id, ) -from ispypsa.templater.mappings import _GENERATORS_EXISTING_PLANNED_PROPERTY_MAP +from ispypsa.templater.mappings import ( + _BATTERY_EXISTING_PLANNED_TECH_PROPERTY_MAP, + _GENERATORS_EXISTING_PLANNED_PROPERTY_MAP, + _PHES_EXISTING_PLANNED_STATION_PROPERTY_MAP, + _STORAGE_EXISTING_PLANNED_UNIT_PROPERTY_MAP, +) # Source (IASR existing_committed_anticipated_additional_generator_summary) column # names → schema output column names. @@ -79,6 +92,20 @@ "minimum_load", ] +_STORAGE_COLUMNS = [ + "name", + "power_station", + "technology", + "geo_id", + "fuel_type", + "capacity", + "storage_capacity", + "efficiency_charge", + "efficiency_discharge", + "commissioning_date", + "closure_year", +] + _COMMISSIONING_DATE_SCHEMA_FORMAT = "%d/%m/%Y" # minimum_load property table spec: @@ -94,11 +121,6 @@ value_col="Min Stable Level (MW)", ) -# The PHES properties table keys Borumba by its short project name; the summary lists -# it under its full project name. TODO: implement as a 'known_value_replacement' when -# storage property merge is implemented. -_BORUMBA_FULL_NAME_MAP = {"Borumba": "QEJP - Borumba"} - # Tumut 3 has a real pump/non-pump unit split that the summary and phes_properties # tables can't currently be reconciled on by name. _validate_phes_routing tolerates # this as known unmatched (rather than renaming units to assign as PHES) as part of @@ -106,15 +128,23 @@ # See Open-ISP/ISPyPSA#131 comment thread. _KNOWN_UNMATCHED_PHES_STATIONS = {"Lower Tumut"} -# Case mismatch between maximum_capacity's IASR ID and the summary's: the 'safe' -# fuzzy-matching threshold (90) would miss (fuzz.ratio("KiataWF1", "KIATAWF1") == 50). -# TODO: rename to use a generic name like IASR_TYPO_FIXES per comment on #143. -_MAXIMUM_CAPACITY_ID_TYPO_FIX = dict( - table_name="maximum_capacity_existing_committed_anticipated_additional_generators", - column="IASR ID", - replacements={"KiataWF1": "KIATAWF1"}, -) +_IASR_TABLE_REPLACEMENTS = [ + # The PHES properties table keys Borumba by its short project name; the summary lists + # it under its full project name. This is a mapping convenience/consistency fix + dict( + table_name="pumped_hydro_existing_committed_anticipated_additional_properties", + column="Power Station", + replacements={"Borumba": "QEJP - Borumba"}, + ), + # Case mismatch between maximum_capacity's IASR ID and the summary's: the 'safe' + # fuzzy-matching threshold (90) would miss (fuzz.ratio("KiataWF1", "KIATAWF1") == 50). + dict( + table_name="maximum_capacity_existing_committed_anticipated_additional_generators", + column="IASR ID", + replacements={"KiataWF1": "KIATAWF1"}, + ), +] # --- public orchestrators --- @@ -159,6 +189,7 @@ def _template_generators_existing_planned( BW01 Bayswater NSW 660.0 NaN # CNSW -> NSW via sub_regional_geography """ logging.info("Creating a template for existing and planned generators") + iasr_tables = _apply_iasr_table_replacements(iasr_tables, _IASR_TABLE_REPLACEMENTS) summary = iasr_tables["existing_committed_anticipated_additional_generator_summary"] phes_properties = iasr_tables[ "pumped_hydro_existing_committed_anticipated_additional_properties" @@ -175,9 +206,9 @@ def _template_generators_existing_planned( non_generator_names = set(summary.loc[is_storage, "name"]) generators = _merge_unit_keyed_properties( generators, - _apply_known_value_replacement(iasr_tables, _MAXIMUM_CAPACITY_ID_TYPO_FIX), + iasr_tables, _GENERATORS_EXISTING_PLANNED_PROPERTY_MAP, - non_generator_names, + exclude_unit_keys=non_generator_names, ) generators = _format_commissioning_date(generators) generators = _merge_minimum_load(generators, iasr_tables) @@ -186,35 +217,76 @@ def _template_generators_existing_planned( def _template_storage_existing_planned( iasr_tables: dict[str, pd.DataFrame], + regional_granularity: str, + sub_regional_geography: pd.DataFrame, ) -> pd.DataFrame: """Templates the existing and planned (ECAA) storage table from the IASR summary. - Currently just the generator/storage split and spine rename — TODO add properties. - Args: iasr_tables: IASR tables; uses - existing_committed_anticipated_additional_generator_summary and - pumped_hydro_existing_committed_anticipated_additional_properties. + existing_committed_anticipated_additional_generator_summary, + pumped_hydro_existing_committed_anticipated_additional_properties, + maximum_capacity_existing_committed_anticipated_additional_generators, + expected_closure_years and battery_properties. + regional_granularity: "sub_regions", "nem_regions", or "single_region". + sub_regional_geography: network_geography templated at "sub_regions" + granularity; columns used: 'geo_id', 'geo_type', 'region_id'. + + I/O Example (subset of columns): + existing_committed_anticipated_additional_generator_summary (abbr.): + IASR ID / DLT names Power Station REZ ID Sub-region Fuel type + BW01 Bayswater NA CNSW Coal + Liddell BESS Liddell BESS N9 CNSW Battery + W/HOE#1 Wivenhoe NA SQ Water + + iasr_tables: + battery_properties: + Technology Charge efficiency_% Discharge efficiency_% + Battery Storage (4hrs storage) 92.5 92.5 + + pumped_hydro_existing_committed_anticipated_additional_properties: + Power Station Pumping efficiency (%) + Wivenhoe 81.0 + + ... plus the other tables in _STORAGE_EXISTING_PLANNED_UNIT_PROPERTY_MAP + + regional_granularity: "nem_regions" - I/O Example (spine columns shown; every other summary column passes through - unchanged): - existing_committed_anticipated_additional_generator_summary: - IASR ID / DLT names Power Station Technology Type - BW01 Bayswater Steam Sub Critical # generator, dropped - Q8 Battery - 2h Q8 Battery Battery Storage (2hrs storage) + sub_regional_geography: + geo_id geo_type region_id + CNSW subregion NSW + SQ subregion QLD returns: - name power_station technology - Q8 Battery - 2h Q8 Battery Battery Storage (2hrs storage) + name power_station geo_id fuel_type efficiency_charge efficiency_discharge + Liddell BESS Liddell BESS N9 Battery 92.5 92.5 + W/HOE#1 Wivenhoe QLD Water 90.0 90.0 """ + logging.info("Creating a template for existing and planned storage") + iasr_tables = _apply_iasr_table_replacements(iasr_tables, _IASR_TABLE_REPLACEMENTS) summary = iasr_tables["existing_committed_anticipated_additional_generator_summary"] phes_properties = iasr_tables[ "pumped_hydro_existing_committed_anticipated_additional_properties" ] is_storage = _is_existing_planned_storage_row(summary, phes_properties) + summary = summary.rename(columns=_SUMMARY_COLUMN_RENAMES) + storage = summary[is_storage].copy() - return storage.rename(columns=_SUMMARY_COLUMN_RENAMES) + storage = _set_geo_id(storage) + storage["geo_id"] = _map_geo_id_to_granularity( + storage["geo_id"], regional_granularity, sub_regional_geography + ) + non_storage_names = set(summary.loc[~is_storage, "name"]) + storage = _merge_unit_keyed_properties( + storage, + iasr_tables, + _STORAGE_EXISTING_PLANNED_UNIT_PROPERTY_MAP, + exclude_unit_keys=non_storage_names, + ) + storage = _merge_storage_type_split_properties(storage, iasr_tables) + storage = _format_commissioning_date(storage) + return storage[_STORAGE_COLUMNS] # --- generator/storage split --- @@ -260,8 +332,8 @@ def _validate_phes_routing( ) -> None: """Raises if a phes_properties station's name doesn't exist anywhere in the summary. - Checked after correcting the known Borumba name mismatch - (``_BORUMBA_FULL_NAME_MAP``) and excusing the one known, documented gap + Checked after correcting the known Borumba name mismatch (see + ``_apply_iasr_table_replacements``) and excusing the one known, documented gap (``_KNOWN_UNMATCHED_PHES_STATIONS`` — Tumut 3's "Lower Tumut"). Any other unmatched name means a real PHES station would otherwise be silently misclassified as a generator. @@ -276,7 +348,7 @@ def _validate_phes_routing( phes_properties: Power Station Wivenhoe # matches - Borumba # matches after _BORUMBA_FULL_NAME_MAP + QEJP - Borumba # already 'fixed' (see _IASR_TABLE_REPLACEMENTS) Lower Tumut # excused by _KNOWN_UNMATCHED_PHES_STATIONS -> no error @@ -286,9 +358,7 @@ def _validate_phes_routing( ['Some New Station'] """ - phes_stations = set( - phes_properties["Power Station"].replace(_BORUMBA_FULL_NAME_MAP) - ) + phes_stations = set(phes_properties["Power Station"]) known_stations = set(summary["Power Station"]) | _KNOWN_UNMATCHED_PHES_STATIONS unmatched = phes_stations - known_stations if unmatched: @@ -301,12 +371,12 @@ def _validate_phes_routing( def _merge_unit_keyed_properties( - generators: pd.DataFrame, + summary: pd.DataFrame, iasr_tables: dict[str, pd.DataFrame], property_map: dict[str, dict], - exclude_unit_keys: set[str] = set(), + exclude_unit_keys: set[str], ) -> pd.DataFrame: - """Merges every property in ``property_map`` onto ``generators``, keyed on IASR ID = name. + """Merges every property in ``property_map`` onto ``summary``, keyed on IASR ID = name. Groups properties by source (table, key_col) — see ``_group_properties_by_source`` — so a table contributing several columns (e.g. maximum_capacity_... feeds @@ -316,7 +386,7 @@ def _merge_unit_keyed_properties( values are mapped. I/O Example: - generators: + summary: name power_station BW01 Bayswater HUNTER1 Hunter Power Station @@ -353,16 +423,7 @@ def _merge_unit_keyed_properties( BW01 Bayswater 660.0 NaN 10.05 HUNTER1 Hunter Power Station 375.0 2025-08-01 10.93 """ - if generators.empty: - # Make sure all expected columns still get added - # Leaving defensive check for empty df ATM -> because it's a subset of a - # templater input table that **could** be empty after splitting. See comments - # on #143. - return generators.assign( - **{new_col: pd.Series(dtype="object") for new_col in property_map} - ) - - generators = generators.copy() + summary = summary.copy() for (table_name, key_col), props in _group_properties_by_source( property_map ).items(): @@ -374,19 +435,19 @@ def _merge_unit_keyed_properties( f"{sorted(props.keys())}", ) resolved_keys = _resolve_unit_keys( - generators["name"], table[key_col], table_name, exclude_unit_keys + summary["name"], table[key_col], table_name, exclude_unit_keys ) for new_col, attrs in props.items(): property_values = _get_property_value_map(table, attrs) - generators[new_col] = resolved_keys.map(property_values) - return generators + summary[new_col] = resolved_keys.map(property_values) + return summary def _resolve_unit_keys( names: pd.Series, table_keys: pd.Series, table_name: str, - exclude_unit_keys: set[str] = set(), + exclude_unit_keys: set[str], ) -> pd.Series: """Fuzzy-resolves ``names`` to ``table_keys``' strings; raises on any miss. @@ -395,7 +456,7 @@ def _resolve_unit_keys( ``table_keys`` is a lookup pool, so its order is irrelevant; the result carries one value per name, in ``names``' order, spelled as in ``table_keys``. A set of unit keys (names) that are known to be out of scope for a given property - merge can be passed to tighten the fuzzy-matching. + merge are passed as `exclude_unit_keys` to tighten the fuzzy-matching. Raises: ValueError: if any unit has no plausible match (above a fuzz ratio threshold @@ -419,22 +480,80 @@ def _resolve_unit_keys( unmatched = resolved[~resolved.isin(table_keys_minus_exclusions)] if not unmatched.empty: raise ValueError( - f"'{table_name}' table missing a row for generator(s): {sorted(unmatched)}" + f"'{table_name}' table missing a row for unit(s): {sorted(unmatched)}" ) return resolved -def _format_commissioning_date(generators: pd.DataFrame) -> pd.DataFrame: +def _merge_storage_type_split_properties( + storage: pd.DataFrame, iasr_tables: dict[str, pd.DataFrame] +) -> pd.DataFrame: + """Merges technology-specific existing/planned storage properties into a summary + table. + + This function splits an input ``storage`` summary table into battery and non-battery + storage units (non-battery is currently PHES-only), merges technology-specific + property values into the corresponding dataframe (see ``_merge_category_keyed_properties``), + applies technology-specific transforms (see ``_derive_phes_symmetric_efficiency``), + and returns a recombined all-storage summary table. Row order is not preserved + by this function, instead (where they each exist) battery unit rows are returned + above PHES unit rows due to the split-then-concat approach. + + I/O Example: + storage (abbr.): + name power_station technology ... + W/HOE#1 Wivenhoe Hydro + QEJP - Borumba QEJP - Borumba Pumped Hydro (24hrs storage) + Liddell BESS Liddell BESS Battery Storage (4hrs storage) + + iasr_tables: + battery_properties: + Technology Charge efficiency_% Discharge efficiency_% + Battery Storage (4hrs storage) 92.5 92.5 + + pumped_hydro_existing_committed_anticipated_additional_properties: + Power Station Pumping efficiency (%) + Wivenhoe 81.0 + QEJP - Borumba 81.0 + + returns: + name power_station technology ... efficiency_charge efficiency_discharge + W/HOE#1 Wivenhoe Hydro 90.0 90.0 + QEJP - Borumba QEJP - Borumba Pumped Hydro (24hrs storage) 90.0 90.0 + Liddell BESS Liddell BESS Battery Storage (4hrs storage) 92.5 92.5 + """ + is_battery = _is_battery_row(storage, col_to_check="technology") + battery_only = storage[is_battery].copy() + phes_only = storage[~is_battery].copy() + + battery_only = _merge_category_keyed_properties( + battery_only, + iasr_tables, + _BATTERY_EXISTING_PLANNED_TECH_PROPERTY_MAP, + df_key_col="technology", + ) + phes_only = _merge_category_keyed_properties( + phes_only, + iasr_tables, + _PHES_EXISTING_PLANNED_STATION_PROPERTY_MAP, + df_key_col="power_station", + ) + phes_only = _derive_phes_symmetric_efficiency(phes_only) + return pd.concat([battery_only, phes_only], axis=0, ignore_index=True) + + +def _format_commissioning_date(summary: pd.DataFrame) -> pd.DataFrame: """Reformats commissioning_date from the IASR's ISO string to the schema's %d/%m/%Y.""" - generators = generators.copy() - generators["commissioning_date"] = pd.to_datetime( - generators["commissioning_date"] + summary = summary.copy() + summary["commissioning_date"] = pd.to_datetime( + summary["commissioning_date"] ).dt.strftime(_COMMISSIONING_DATE_SCHEMA_FORMAT) - return generators + return summary def _merge_minimum_load( - generators: pd.DataFrame, iasr_tables: dict[str, pd.DataFrame] + generators: pd.DataFrame, + iasr_tables: dict[str, pd.DataFrame], ) -> pd.DataFrame: """Merges technology-specific minimum_load property for coal and gas generators. @@ -464,8 +583,6 @@ def _merge_minimum_load( ANGAS1 Reciprocating engine 3.0 Q1G1 Large scale Solar PV NaN # neither coal nor gas """ - if generators.empty: - return generators.assign(minimum_load=pd.Series(dtype="float64")) generators = generators.copy() generators["minimum_load"] = float("nan") @@ -525,6 +642,7 @@ def _merge_minimum_load_property( generators[is_candidate]["name"], table[property_spec["key_col"]], property_spec["table"], + exclude_unit_keys=set(), ) values = _get_property_value_map(table, property_spec) generators.loc[is_candidate, "minimum_load"] = resolved_keys.map(values) diff --git a/src/ispypsa/templater/helpers.py b/src/ispypsa/templater/helpers.py index 75caea83..48d2e6ed 100644 --- a/src/ispypsa/templater/helpers.py +++ b/src/ispypsa/templater/helpers.py @@ -552,43 +552,46 @@ def _assert_table_valid( raise ValueError(f"'{table_name}' table is empty - cannot merge {merge_desc}") -def _apply_known_value_replacement( - iasr_tables: dict[str, pd.DataFrame], correction: dict +def _apply_iasr_table_replacements( + iasr_tables: dict[str, pd.DataFrame], corrections: list[dict] ) -> dict[str, pd.DataFrame]: - """Returns ``iasr_tables`` with a known correction applied to one table's column. + """Returns ``iasr_tables`` with 'corrections' applied to input IASR tables. - Shared shape for a small, explicitly declared fix (a documented typo or naming - mismatch) to a single column of a single source table. ``correction`` bundles the - fix's specifics (``table_name``, ``column``, ``replacements``). Returns a shallow - copy of ``iasr_tables`` with only that table replaced. + Shared shape for small, explicitly declared fixes (a documented typo or naming + mismatch) to specified locations. ``corrections`` can carry multiple fixes, + each with 'fix' specifics (``table_name``, ``column``, ``replacements``) bundled. + Returns a shallow copy of ``iasr_tables`` with only listed tables replaced. Note: while fuzzy-matching is used to standardise names or other ID strings, some typos/diffs are too 'big' to pass any safe fuzzy-match threshold (see example below - fuzz.ratio("KiataWF1", "KIATAWF1") == 50). This function explicitly handles those known instances where this is the case. - I/O Example (correction = existing_planned._MAXIMUM_CAPACITY_ID_TYPO_FIX): + I/O Example: iasr_tables["maximum_capacity_..."]: IASR ID Power Station Installed capacity (MW) KiataWF1 Kiata Wind Farm 31.05 BW01 Bayswater 660.0 - correction: + corrections (as a single dict element in list): table_name: "maximum_capacity_..." column: "IASR ID" replacements: {"KiataWF1": "KIATAWF1"} - returns copy of iasr_tables with only that one table edited: + returns copy of iasr_tables with only listed tables edited: iasr_tables["maximum_capacity_..."]: IASR ID Power Station Installed capacity (MW) KIATAWF1 Kiata Wind Farm 31.05 BW01 Bayswater 660.0 """ - table_name = correction["table_name"] - corrected = iasr_tables[table_name].replace( - {correction["column"]: correction["replacements"]} - ) - return {**iasr_tables, table_name: corrected} + corrected_tables = iasr_tables + for correction in corrections: + table_name = correction["table_name"] + col_to_fix = correction["column"] + replacements = correction["replacements"] + corrected = iasr_tables[table_name].replace({col_to_fix: replacements}) + corrected_tables = {**corrected_tables, table_name: corrected} + return corrected_tables def _group_properties_by_source( @@ -596,7 +599,7 @@ def _group_properties_by_source( ) -> dict[tuple[str, str], dict]: """Groups a property map's entries by their source (table, key_col). - Shared by ``new_entrants._merge_properties`` and + Shared by ``_merge_category_keyed_properties`` and ``existing_planned._merge_unit_keyed_properties`` so a table contributing several properties (e.g. ``battery_properties`` feeds six) is validated and key-resolved once per source, not once per property. @@ -673,10 +676,68 @@ def _get_property_value_map( attrs.get("scale", 1.0) ) # TODO: 'year' type cols become floats from this transform - leave for - # validator to type-correct or edit handling here? + # validator to type-correct or edit handling here? See Open-ISP/ISPyPSA#145 return value_map +def _merge_category_keyed_properties( + df: pd.DataFrame, + iasr_tables: dict[str, pd.DataFrame], + property_map: dict[str, dict], + df_key_col: str, +) -> pd.DataFrame: + """Merges every non-unit-keyed property in ``property_map`` onto ``df``. + + Groups properties by their source (table, key_col) — see + ``_group_properties_by_source`` — so a table that contributes several properties + (e.g. ``battery_properties`` feeds six new entrant storage properties) is + validated and fuzzy-matched against ``df_key_col`` values once per property map. + + I/O Example: + An abbreviated example merging the 'efficiency_charge' property into an + 'existing_planned_storage' summary table. + + df: + name technology + Liddell BESS Battery storage (4hrs storage) + + df_key_col = "technology" + + property_map: + efficiency_charge: table="battery_properties", + key_col="Technology", + value_col="Charge efficiency_%" + + iasr_tables['battery_properties']: + Technology Charge efficiency_% + Battery storage (4hrs storage) 92.5 + + returns (adds one column per map key): + name technology efficiency_charge + Liddell BESS Battery storage (4hrs storage) 92.5 + """ + df = df.copy() + for (table_name, key_col), props in _group_properties_by_source( + property_map + ).items(): + table = iasr_tables[table_name] + _assert_table_valid( + table, + table_name, + _required_property_columns(props), + f"{sorted(props.keys())}", + ) + matched_key_col = _fuzzy_map_to_allowed_values( + df[df_key_col], + table[key_col], + task_desc=f"merging properties from '{table_name}'", + ) + for new_col, attrs in props.items(): + property_values = _get_property_value_map(table, attrs) + df[new_col] = matched_key_col.map(property_values) + return df + + def _is_battery_row( df: pd.DataFrame, col_to_check: str = "Technology Type" ) -> pd.Series: @@ -714,7 +775,9 @@ def _derive_phes_symmetric_efficiency(phes: pd.DataFrame) -> pd.DataFrame: The IASR PHES tables give only a single round-trip efficiency. Assuming symmetric legs, each one-way efficiency is its square root, so e.g. a 76% round trip becomes - ~87.2% charge and ~87.2% discharge (sqrt(0.76) ≈ 0.872). + ~87.2% charge and ~87.2% discharge (sqrt(0.76) ≈ 0.872). The function returns + the input `phes` df with two new columns (efficiency_charge and efficiency_discharge), + dropping the intermediate round_trip_efficiency column. I/O Example: phes: @@ -722,14 +785,14 @@ def _derive_phes_symmetric_efficiency(phes: pd.DataFrame) -> pd.DataFrame: NQ Pumped Hydro-10h 76.0 returns (adds the two efficiency columns): - name round_trip_efficiency efficiency_charge efficiency_discharge - NQ Pumped Hydro-10h 76.0 87.18 87.18 + name efficiency_charge efficiency_discharge + NQ Pumped Hydro-10h 87.18 87.18 """ phes = phes.copy() one_way_efficiency = (phes["round_trip_efficiency"] / 100) ** 0.5 * 100 phes["efficiency_charge"] = one_way_efficiency phes["efficiency_discharge"] = one_way_efficiency - return phes + return phes.drop(columns=["round_trip_efficiency"]) def _standardise_storage_capitalisation(series: pd.Series) -> pd.Series: diff --git a/src/ispypsa/templater/mappings.py b/src/ispypsa/templater/mappings.py index a4c5f245..ae6222ec 100644 --- a/src/ispypsa/templater/mappings.py +++ b/src/ispypsa/templater/mappings.py @@ -666,7 +666,7 @@ """ New entrant property columns (keys) mapped to the IASR table and columns that contain property values and the technology for which the values apply. Consumed by -``ispypsa.templater.new_entrants`` via ``_merge_properties``. +``ispypsa.templater.new_entrants`` via ``_merge_category_keyed_properties``. `table`: IASR table name holding the named property (key) `key_col`: column in the IASR table that contains the 'technology' string. @@ -766,12 +766,12 @@ } """ -Existing/planned (ECAA) generator property columns (keys) mapped to the IASR table and -column that contains their values. Consumed by +Existing/planned (ECAA) generator and storage unit-level property columns (keys) mapped +to the IASR table and column that contains their values. Consumed by ``ispypsa.templater.existing_planned`` via ``_merge_unit_keyed_properties``. Shaped like the new entrant maps above, but every entry here shares the same -``key_col`` (``IASR ID``) — each generator's ``name`` is resolved against it via +``key_col`` (``IASR ID``) — each unit's ``name`` is resolved against it via fuzzy matching (small typos only; see ``existing_planned._resolve_unit_keys``) rather than the technology-level fuzzy grouping the new entrant maps need. @@ -815,3 +815,62 @@ value_col="Expected Closure Year (Calendar year)", ), } + +_STORAGE_EXISTING_PLANNED_UNIT_PROPERTY_MAP = { + "capacity": dict( + table="maximum_capacity_existing_committed_anticipated_additional_generators", + key_col="IASR ID", + value_col="Installed capacity (MW)", + ), + "storage_capacity": dict( + table="maximum_capacity_existing_committed_anticipated_additional_generators", + key_col="IASR ID", + value_col="Storage Capacity (MWh)", + ), + "commissioning_date": dict( + table="maximum_capacity_existing_committed_anticipated_additional_generators", + key_col="IASR ID", + value_col="Commissioning date", + numeric=False, + ), + "closure_year": dict( + table="expected_closure_years", + key_col="IASR ID", + value_col="Expected Closure Year (Calendar year)", + ), +} + +""" +Existing/planned (ECAA) storage properties that aren't published per unit, mapped to +the IASR table and column that contains their values. Consumed by +``ispypsa.templater.existing_planned`` via ``_merge_storage_type_split_properties``, +which merges each map onto its own subset of storage rows with +``helpers._merge_category_keyed_properties``. + +Each map is keyed on a different category shared by many units: + - batteries on ``technology`` (``battery_properties``' ``Technology``) + - PHES on ``power_station`` (the PHES properties table's ``Power Station``) + +Entries use the same fields as the unit-level maps above. +""" + +_BATTERY_EXISTING_PLANNED_TECH_PROPERTY_MAP = { + "efficiency_charge": dict( + table="battery_properties", + key_col="Technology", + value_col="Charge efficiency_%", + ), + "efficiency_discharge": dict( + table="battery_properties", + key_col="Technology", + value_col="Discharge efficiency_%", + ), +} + +_PHES_EXISTING_PLANNED_STATION_PROPERTY_MAP = { + "round_trip_efficiency": dict( + table="pumped_hydro_existing_committed_anticipated_additional_properties", + key_col="Power Station", + value_col="Pumping efficiency (%)", + ), +} diff --git a/src/ispypsa/templater/new_entrants.py b/src/ispypsa/templater/new_entrants.py index 2f0ed837..9fc777c4 100644 --- a/src/ispypsa/templater/new_entrants.py +++ b/src/ispypsa/templater/new_entrants.py @@ -11,8 +11,9 @@ 3. Derives geo_id (REZ ID or sub-region) 4. (Generators only) Derives resource_type from the VRE code in the IASR ID 5. Merges in per-technology property values — each a single number looked up by - technology, via _merge_properties (see the property merge maps in mappings.py, - e.g. _GENERATORS_NEW_ENTRANT_PROPERTY_MAP). Generators and storage share a common + technology, via helpers._merge_category_keyed_properties (see the property + merge maps in mappings.py, e.g. _GENERATORS_NEW_ENTRANT_PROPERTY_MAP). + Generators and storage share a common set of these (_COMMON_NEW_ENTRANT_PROPERTY_MAP). Storage additionally splits into battery and pumped-hydro (PHES) rows, which take their storage-specific properties from different IASR tables, then recombines them before merging @@ -35,18 +36,16 @@ import pandas as pd from ispypsa.templater.helpers import ( - _apply_known_value_replacement, + _apply_iasr_table_replacements, _assert_table_valid, _derive_phes_symmetric_efficiency, _fuzzy_map_to_allowed_values, - _get_property_value_map, - _group_properties_by_source, _is_battery_row, _is_pumped_hydro_row, _is_storage_row, _is_subregion_geo_id, _map_geo_id_to_granularity, - _required_property_columns, + _merge_category_keyed_properties, _set_geo_id, ) from ispypsa.templater.mappings import ( @@ -207,7 +206,9 @@ def _template_generators_new_entrant( gens = gens.rename(columns=_SUMMARY_COLUMN_RENAMES) gens = _set_geo_id(gens) gens = _add_resource_type(gens) - gens = _merge_properties(gens, iasr_tables, _GENERATORS_NEW_ENTRANT_PROPERTY_MAP) + gens = _merge_category_keyed_properties( + gens, iasr_tables, _GENERATORS_NEW_ENTRANT_PROPERTY_MAP, df_key_col="technology" + ) _assert_build_cost_zone_matches_geo_id(gens) gens = _merge_lcf_build(gens, iasr_tables["technology_specific_lcfs"]) gens = _merge_lcf_om(gens, iasr_tables["locational_cost_factors"]) @@ -261,16 +262,22 @@ def _template_storage_new_entrant( storage = new_entrants_summary[_is_storage_row(new_entrants_summary)].copy() storage = storage.rename(columns=_SUMMARY_COLUMN_RENAMES) storage = _set_geo_id(storage) - batteries = _merge_properties( + batteries = _merge_category_keyed_properties( storage[_is_battery_row(storage, col_to_check="technology")], iasr_tables, _STORAGE_BATTERY_PROPERTY_MAP, + df_key_col="technology", ) phes = _merge_phes_properties( storage[_is_pumped_hydro_row(storage, col_to_check="technology")], iasr_tables ) storage = pd.concat([batteries, phes], ignore_index=True) - storage = _merge_properties(storage, iasr_tables, _COMMON_NEW_ENTRANT_PROPERTY_MAP) + storage = _merge_category_keyed_properties( + storage, + iasr_tables, + _COMMON_NEW_ENTRANT_PROPERTY_MAP, + df_key_col="technology", + ) _assert_build_cost_zone_matches_geo_id(storage) storage = _merge_lcf_build(storage, iasr_tables["technology_specific_lcfs"]) storage = _merge_lcf_om(storage, iasr_tables["locational_cost_factors"]) @@ -284,52 +291,6 @@ def _template_storage_new_entrant( ) -# --- shared helpers --- - - -def _merge_properties( - new_entrants: pd.DataFrame, - iasr_tables: dict[str, pd.DataFrame], - property_map: dict[str, dict], -) -> pd.DataFrame: - """Merges every property in ``property_map`` onto ``new_entrants``. - - Groups properties by their source (table, key_col) — see - ``_group_properties_by_source`` — so a table that contributes several properties - (e.g. ``battery_properties`` feeds six) is validated and fuzzy-matched against - ``new_entrants``' 'technology' once per property map. - - I/O Example (property_map = _STORAGE_BATTERY_PROPERTY_MAP, abbreviated): - new_entrants: - name technology - NQ Battery - 2h Battery Storage (2hrs storage) - - returns (adds one column per map key): - name technology storage_hours efficiency_charge ... - NQ Battery - 2h Battery Storage (2hrs storage) 2.0 92.0 ... - """ - new_entrants = new_entrants.copy() - for (table_name, key_col), props in _group_properties_by_source( - property_map - ).items(): - table = iasr_tables[table_name] - _assert_table_valid( - table, - table_name, - _required_property_columns(props), - f"{sorted(props.keys())}", - ) - matched_technology = _fuzzy_map_to_allowed_values( - new_entrants["technology"], - table[key_col], - task_desc=f"merging new entrant properties from '{table_name}'", - ) - for new_col, attrs in props.items(): - property_values = _get_property_value_map(table, attrs) - new_entrants[new_col] = matched_technology.map(property_values) - return new_entrants - - # --- regional granularity collapse --- @@ -642,10 +603,11 @@ def _merge_phes_properties( """ phes = phes.copy() phes["technology"] = _override_botn_technology(phes) - phes = _merge_properties( + phes = _merge_category_keyed_properties( phes, - _apply_known_value_replacement(iasr_tables, _PHES_BOTN_KEY_FIX), + _apply_iasr_table_replacements(iasr_tables, [_PHES_BOTN_KEY_FIX]), _STORAGE_PHES_PROPERTY_MAP, + df_key_col="technology", ) phes = _derive_phes_symmetric_efficiency(phes) return phes diff --git a/tests/test_cli/test_create_ispypsa_inputs_new_table_formats.py b/tests/test_cli/test_create_ispypsa_inputs_new_table_formats.py index da69f587..815d705c 100644 --- a/tests/test_cli/test_create_ispypsa_inputs_new_table_formats.py +++ b/tests/test_cli/test_create_ispypsa_inputs_new_table_formats.py @@ -315,6 +315,10 @@ def test_create_ispypsa_inputs_task_new_format( # new entrant), so this is the same at every granularity. _EXPECTED_GENERATORS_EXISTING_PLANNED_ROWS_75 = 531 +# As above - so far no 'Power Station' level collapse implemented so this count +# remains the same at every granularity. +_EXPECTED_STORAGE_EXISTING_PLANNED_ROWS_75 = 111 + # REZ sub-zone ids used by some existing/planned generators that don't appear in # renewable_energy_zones (only their parent REZ does, e.g. "Q8" not "Q8a"/"Q8b"/ # "Q8c"). Same gap class as _NON_REZ_PLACEHOLDER_GEO_IDS (#133), different cause. @@ -375,6 +379,7 @@ def test_create_ispypsa_inputs_new_format( gens_new_entrant = pd.read_csv(output_dir / "generators_new_entrant.csv") storage_new_entrant = pd.read_csv(output_dir / "storage_new_entrant.csv") gens_existing_planned = pd.read_csv(output_dir / "generators_existing_planned.csv") + storage_existing_planned = pd.read_csv(output_dir / "storage_existing_planned.csv") # network_geography — one row per (sub-)region or REZ; geo_ids are unique. assert len(geo) == _GEOS_PER_GRANULARITY_75[granularity] + _NUM_REZS_75 @@ -443,6 +448,16 @@ def test_create_ispypsa_inputs_new_format( set(geo["geo_id"]) | _NON_REZ_PLACEHOLDER_GEO_IDS | _MISSING_REZ_SUBZONE_GEO_IDS ) + # as above - storage_existing_planned — one row per existing/planned storage unit, + # no collapse step so the row count doesn't vary. geo_ids are real network_geography + # entries, except the two pre-existing gaps noted above ("Non-REZ" placeholders + # and REZ sub-zones). + assert len(storage_existing_planned) == _EXPECTED_STORAGE_EXISTING_PLANNED_ROWS_75 + assert storage_existing_planned["name"].is_unique + assert set(storage_existing_planned["geo_id"]) <= ( + set(geo["geo_id"]) | _NON_REZ_PLACEHOLDER_GEO_IDS | _MISSING_REZ_SUBZONE_GEO_IDS + ) + # costs_connection - every new entrant technology + geo_id + year has a row # geo_id's are real network_geography entries + _NON_REZ_PLACEHOLDER_GEO_IDS assert len(costs_connection) == _EXPECTED_COSTS_CONNECTION_ROWS_75[granularity] diff --git a/tests/test_templater/test_create_ispypsa_inputs_template.py b/tests/test_templater/test_create_ispypsa_inputs_template.py index 5d0bee8c..62729baf 100644 --- a/tests/test_templater/test_create_ispypsa_inputs_template.py +++ b/tests/test_templater/test_create_ispypsa_inputs_template.py @@ -74,7 +74,7 @@ def _new_entrant_property_tables(csv_str_to_df) -> dict[str, pd.DataFrame]: Wind, 20.0, $ Large scale Solar PV, 15.0, $ OCGT (small GT), 17.0, $ - Battery Storage (2hrs storage), 13.5, $ + Battery storage (2hrs storage), 13.5, $ Pumped Hydro (24hrs storage), 78.5, $ BOTN - Cethana, 78.5, $ """), @@ -89,7 +89,7 @@ def _new_entrant_property_tables(csv_str_to_df) -> dict[str, pd.DataFrame]: Wind, 5, 30 Large scale Solar PV, 25, 30 OCGT (small GT), 25, 40 - Battery Storage (2hrs storage), 20, 20 + Battery storage (2hrs storage), 20, 20 Pumped Hydro (24hrs storage), 40, 90 BOTN - Cethana, 40, 90 """), @@ -104,7 +104,7 @@ def _new_entrant_property_tables(csv_str_to_df) -> dict[str, pd.DataFrame]: Wind, 0.0 Large scale Solar PV, 0.0 OCGT (small GT), 50.0 - Battery Storage (2hrs storage), 0.0 + Battery storage (2hrs storage), 0.0 Pumped Hydro (24hrs storage), 40.0 BOTN - Cethana, 40.0 """), @@ -118,7 +118,7 @@ def _new_entrant_property_tables(csv_str_to_df) -> dict[str, pd.DataFrame]: BOTN - Cethana - 20h, 20, 80 """), "technology_specific_lcfs": csv_str_to_df(""" - Cost zone / REZ ID, REZ name / Description, Wind, Large scale Solar PV, OCGT (small GT), Battery Storage (2hrs storage), Pumped Hydro (24hrs storage), BOTN - Cethana + Cost zone / REZ ID, REZ name / Description, Wind, Large scale Solar PV, OCGT (small GT), Battery storage (2hrs storage), Pumped Hydro (24hrs storage), BOTN - Cethana Q1, Far North QLD, 1.05, 1.08, Not Applicable, Not Applicable, Not Applicable, Not Applicable CNSW, Subregional Ref Node, Not Applicable, Not Applicable, 1.04, 1.03, Not Applicable, Not Applicable SNW, Subregional Ref Node, Not Applicable, Not Applicable, 1.00, 1.01, Not Applicable, Not Applicable @@ -141,29 +141,32 @@ def _new_entrant_property_tables(csv_str_to_df) -> dict[str, pd.DataFrame]: def _existing_planned_tables(csv_str_to_df) -> dict[str, pd.DataFrame]: """The ECAA summary plus its per-unit property tables, wiring-only fixture. - BW01 (CNSW coal generator) and Q1G1 (Q1 REZ solar generator) prove the - generator/geo_id wiring at every granularity (CNSW collapses to NSW at - nem_regions/single_region; Q1 stays untouched, matching new entrant's REZ - fixtures above). Q8_BATT_2H proves the storage split still excludes storage - rows from generators_existing_planned's output. Detailed merge behaviour is - covered in test_existing_planned.py; here they just need to be present for - wiring to run. + Detailed merge behaviour is covered in test_existing_planned.py; here they + just need to be present for wiring to run. """ summary = csv_str_to_df(""" IASR ID / DLT names, Power Station, Technology Type, REZ ID, Sub-region, Fuel type, Fuel cost mapping BW01, Bayswater, Steam Sub Critical, Not Applicable, CNSW, Black Coal, Bayswater Q1G1, Solar Farm, Large scale Solar PV, Q1, NQ, Solar, Solar Farm - SAMPLE_BATT, Sample Battery, Battery Storage (2hrs storage), Not Applicable, CNSW, Battery, Sample Battery + W/HOE#1, Wivenhoe, Hydro, Not Applicable, NQ, Water, Hydro + ORANA, Orana BESS, Battery storage (2hrs storage), N3, CNSW, Battery, Battery """) return { "existing_committed_anticipated_additional_generator_summary": summary, - "pumped_hydro_existing_committed_anticipated_additional_properties": csv_str_to_df( - "Power Station" - ), + "pumped_hydro_existing_committed_anticipated_additional_properties": csv_str_to_df(""" + Power Station, Installed capacity (MW),Storage capacity (hours), Pumping efficiency (%) + Wivenhoe, 570, 10.0, 70 + """), "maximum_capacity_existing_committed_anticipated_additional_generators": csv_str_to_df(""" - IASR ID, Installed capacity (MW), Commissioning date - BW01, 660.0, - Q1G1, 100.0, 2028-12-01 + IASR ID, Installed capacity (MW), Storage Capacity (MWh), Commissioning date + BW01, 660.0, , + Q1G1, 100.0, , 2028-12-01 + W/HOE#1, 285.0, 3000.0, + ORANA, 415.0, 1660.0, 2026-06-01 + """), + "battery_properties": csv_str_to_df(""" + Technology, Energy capacity_Hours, Charge efficiency_%, Discharge efficiency_%, Allowable max state of charge_%, Allowable min state of charge_%, Annual degradation_% + Battery storage (2hrs storage), 2.0, 92.0, 92.0, 100, 0, 1.8 """), # Built directly - the real column name's trailing "1," is a literal comma "variable_opex_existing_committed_anticipated_additional_generators": pd.DataFrame( @@ -181,6 +184,8 @@ def _existing_planned_tables(csv_str_to_df) -> dict[str, pd.DataFrame]: IASR ID, Expected Closure Year (Calendar year) BW01, 2033 Q1G1, 2100 + W/HOE#1, 2084 + ORANA, 2066 """), "coal_minimum_stable_level": csv_str_to_df(""" IASR ID, Technology Type, Minimum Stable Level (MW)_Typical Lowest Band @@ -543,8 +548,8 @@ def test_create_ispypsa_inputs_template_new_format(csv_str_to_df): # BOTN - Cethana (Pumped Hydro), the only storage row in the fixture. assert len(storage_new_entrant) == 1 - # generators_existing_planned — BW01 (CNSW) and Q1G1 (Q1 REZ); SAMPLE_BATT - # is excluded (storage) + # generators_existing_planned — BW01 (CNSW) and Q1G1 (Q1 REZ); W/HOE#1 and ORANA + # excluded (storage) generators_existing_planned = result["generators_existing_planned"] assert set(generators_existing_planned.columns) == { "name", @@ -563,6 +568,25 @@ def test_create_ispypsa_inputs_template_new_format(csv_str_to_df): assert set(generators_existing_planned["geo_id"]) == {"CNSW", "Q1"} assert len(generators_existing_planned) == 2 + # storage_existing_planned — W/HOE#1 (NQ) and ORANA (N3); BW01 and Q1G1 + # excluded (generation) + storage_existing_planned = result["storage_existing_planned"] + assert set(storage_existing_planned.columns) == { + "name", + "power_station", + "technology", + "geo_id", + "fuel_type", + "capacity", + "storage_capacity", + "efficiency_charge", + "efficiency_discharge", + "commissioning_date", + "closure_year", + } + assert set(storage_existing_planned["geo_id"]) == {"NQ", "N3"} + assert len(storage_existing_planned) == 2 + # Custom-constraints tables are spliced into the output via # template.update(template_custom_constraints_from_plexos(...)). The # templater is mocked with _stub_custom_constraints_tables (full content is @@ -670,10 +694,10 @@ def test_create_ispypsa_inputs_template_new_format_nem_regions(csv_str_to_df): CNSW OCGT Small, OCGT (small GT), Gas, Not Applicable, CNSW, CNSW SNW OCGT Small, OCGT (small GT), Gas, Not Applicable, SNW, SNW NQ OCGT Small, OCGT (small GT), Gas, Not Applicable, NQ, NQ - CNSW Battery - 2h, Battery Storage (2hrs storage), Battery, Not Applicable, CNSW, CNSW - SNW Battery - 2h, Battery Storage (2hrs storage), Battery, Not Applicable, SNW, SNW - NQ Battery - 2h, Battery Storage (2hrs storage), Battery, Not Applicable, NQ, NQ - Q1 Battery - 2h, Battery Storage (2hrs storage), Battery, Q1, NQ, Q1 + CNSW Battery - 2h, Battery storage (2hrs storage), Battery, Not Applicable, CNSW, CNSW + SNW Battery - 2h, Battery storage (2hrs storage), Battery, Not Applicable, SNW, SNW + NQ Battery - 2h, Battery storage (2hrs storage), Battery, Not Applicable, NQ, NQ + Q1 Battery - 2h, Battery storage (2hrs storage), Battery, Q1, NQ, Q1 """) with ( @@ -781,6 +805,11 @@ def test_create_ispypsa_inputs_template_new_format_nem_regions(csv_str_to_df): assert set(generators_existing_planned["geo_id"]) == {"NSW", "Q1"} assert len(generators_existing_planned) == 2 + # storage_existing_planned - NQ relabels to QLD; N3 REZ stays untouched + storage_existing_planned = result["storage_existing_planned"] + assert set(storage_existing_planned["geo_id"]) == {"QLD", "N3"} + assert len(storage_existing_planned) == 2 + # Not templated at this granularity (see empty_custom_constraint_tables); # the mock turns a regression in that gate into a clean assertion failure # rather than a PLEXOS-extract disk read. @@ -858,10 +887,10 @@ def test_create_ispypsa_inputs_template_new_format_single_region(csv_str_to_df): CNSW OCGT Small, OCGT (small GT), Gas, Not Applicable, CNSW, CNSW SNW OCGT Small, OCGT (small GT), Gas, Not Applicable, SNW, SNW NQ OCGT Small, OCGT (small GT), Gas, Not Applicable, NQ, NQ - CNSW Battery - 2h, Battery Storage (2hrs storage), Battery, Not Applicable, CNSW, CNSW - SNW Battery - 2h, Battery Storage (2hrs storage), Battery, Not Applicable, SNW, SNW - NQ Battery - 2h, Battery Storage (2hrs storage), Battery, Not Applicable, NQ, NQ - Q1 Battery - 2h, Battery Storage (2hrs storage), Battery, Q1, NQ, Q1 + CNSW Battery - 2h, Battery storage (2hrs storage), Battery, Not Applicable, CNSW, CNSW + SNW Battery - 2h, Battery storage (2hrs storage), Battery, Not Applicable, SNW, SNW + NQ Battery - 2h, Battery storage (2hrs storage), Battery, Not Applicable, NQ, NQ + Q1 Battery - 2h, Battery storage (2hrs storage), Battery, Q1, NQ, Q1 """) with ( @@ -965,6 +994,11 @@ def test_create_ispypsa_inputs_template_new_format_single_region(csv_str_to_df): assert set(generators_existing_planned["geo_id"]) == {"NEM", "Q1"} assert len(generators_existing_planned) == 2 + # storage_existing_planned - NQ relabels to NEM; N3 REZ stays untouched + storage_existing_planned = result["storage_existing_planned"] + assert set(storage_existing_planned["geo_id"]) == {"NEM", "N3"} + assert len(storage_existing_planned) == 2 + # Not templated at this granularity (see empty_custom_constraint_tables); # the mock turns a regression in that gate into a clean assertion failure # rather than a PLEXOS-extract disk read. diff --git a/tests/test_templater/test_existing_planned.py b/tests/test_templater/test_existing_planned.py index cb037c9f..fdde63d6 100644 --- a/tests/test_templater/test_existing_planned.py +++ b/tests/test_templater/test_existing_planned.py @@ -5,6 +5,7 @@ _format_commissioning_date, _is_existing_planned_storage_row, _merge_minimum_load, + _merge_storage_type_split_properties, _merge_unit_keyed_properties, _resolve_unit_keys, _template_generators_existing_planned, @@ -20,20 +21,20 @@ def test_is_existing_planned_storage_row(csv_str_to_df): Power Station, Technology Type Wivenhoe, Hydro Tarong, Steam Sub Critical - Q8 Battery, Battery Storage (2hrs storage) + Q8 Battery, Battery storage (2hrs storage) QEJP - Borumba, Pumped Hydro (24hrs storage) """) phes_properties = csv_str_to_df(""" Power Station Wivenhoe - Borumba + QEJP - Borumba """) result = _is_existing_planned_storage_row(summary, phes_properties) # Wivenhoe routes to storage by PHES-table presence despite being labelled plain # "Hydro"; Q8 Battery routes by technology; Tarong is neither; QEJP - Borumba - # routes by technology AND is matched to Borumba in phes_properties. + # routes by technology AND is matched by Power Station expected = pd.Series([True, False, True, True]) pd.testing.assert_series_equal(result, expected) @@ -53,7 +54,7 @@ def test_is_existing_planned_storage_row_empty_phes_properties(csv_str_to_df): summary = csv_str_to_df(""" Power Station, Technology Type Tarong, Steam Sub Critical - Q8 Battery, Battery Storage (2hrs storage) + Q8 Battery, Battery storage (2hrs storage) """) phes_properties = pd.DataFrame(columns=["Power Station"]) @@ -63,21 +64,6 @@ def test_is_existing_planned_storage_row_empty_phes_properties(csv_str_to_df): pd.testing.assert_series_equal(result, expected) -def test_is_existing_planned_storage_row_empty_summary_with_phes_properties( - csv_str_to_df, -): - # An empty summary can't account for any PHES station, so routing validation - # raises rather than silently classifying nothing as storage. - summary = pd.DataFrame(columns=["Power Station", "Technology Type"]) - phes_properties = csv_str_to_df(""" - Power Station - Wivenhoe - """) - - with pytest.raises(ValueError, match=r"\['Wivenhoe'\]"): - _validate_phes_routing(summary, phes_properties) - - # --- _validate_phes_routing --- @@ -85,19 +71,17 @@ def test_validate_phes_routing_tolerates_lower_tumut(csv_str_to_df): summary = csv_str_to_df(""" Power Station, Technology Type Tumut 3, Hydro - QEJP - Borumba, Pumped Hydro (24hrs storage) Tarong, Steam Sub Critical """) phes_properties = csv_str_to_df(""" Power Station Lower Tumut - Borumba """) _validate_phes_routing(summary, phes_properties) # no error -def test_validate_phes_routing_raises_on_unrecognised_station(csv_str_to_df): +def test_validate_phes_routing_raises_on_missing_station(csv_str_to_df): summary = csv_str_to_df(""" Power Station, Technology Type Tumut 3, Hydro @@ -120,16 +104,31 @@ def test_resolve_unit_keys(): # A single-character typo in the table's key is close enough (threshold=90) to # resolve -- the *table's* spelling is returned, but in `names` order. # A non-empty exclude_unit_keys correctly removes known out-of-scope table_keys - # before fuzzy-matching ("BAYSWATER02"). - names = pd.Series(["BW01", "BW02", "BAYSWATER01"]) - table_keys = pd.Series(["BW01", "bAYSWATER01", "BW02", "BAYSWATER02"]) + # before fuzzy-matching ("Another_Very_Long_name1"). + names = pd.Series(["BW01", "BW02", "BAYSWATER01", "Another_Very_Long_name"]) + table_keys = pd.Series( + [ + "BW01", + "bAYSWATER01", + "BW02", + "Another_Very_Long_Name", # No exact match in 'names' (but close) + "Another_Very_Long_name1", # Excluded by 'exclude_unit_keys' + ] + ) - # if "BAYSWATER02" were not excluded it would win the fuzzy-match (incorrectly) - exclude_unit_keys = set(["BAYSWATER02"]) + # if "Another_Very_Long_name1" were not excluded it would win the fuzzy-match (incorrectly) + exclude_unit_keys = set(["Another_Very_Long_name1"]) result = _resolve_unit_keys(names, table_keys, "heat_rates_...", exclude_unit_keys) - expected_result = pd.Series(["BW01", "BW02", "bAYSWATER01"]) + expected_result = pd.Series( + [ + "BW01", + "BW02", + "bAYSWATER01", + "Another_Very_Long_Name", + ] + ) pd.testing.assert_series_equal(result, expected_result) @@ -138,7 +137,7 @@ def test_resolve_unit_keys_raises_on_unmatched(): table_keys = pd.Series(["BW01"]) with pytest.raises(ValueError, match=r"'heat_rates_\.\.\.'.*\['UNKNOWN1'\]"): - _resolve_unit_keys(names, table_keys, "heat_rates_...") + _resolve_unit_keys(names, table_keys, "heat_rates_...", set()) # --- _merge_unit_keyed_properties --- @@ -148,15 +147,18 @@ def test_merge_unit_keyed_properties_single_table_multiple_columns(csv_str_to_df # capacity and commissioning_date both come from the same table - key-resolved once. # Includes small typo fuzzy-match check (EXAMPLE_GEN <-> EXAMPLE_GEn) and non-numeric # values (not coerced). - generators = csv_str_to_df(""" + summary = csv_str_to_df(""" name BW01 EXAMPLE_GEN + Another_Very_Long_name """) maximum_capacity = csv_str_to_df(""" - IASR ID, Installed capacity (MW), Commissioning date - BW01, 660.0, 2028-12-01 - EXAMPLE_GEn, 100.0, + IASR ID, Installed capacity (MW), Commissioning date + BW01, 660.0, 2028-12-01 + EXAMPLE_GEn, 100.0, + Another_Very_Long_Name, 54.0, 2027-05-01 + Another_Very_Long_name1, 540.0, 2027-10-01 """) property_map = { "capacity": { @@ -173,15 +175,75 @@ def test_merge_unit_keyed_properties_single_table_multiple_columns(csv_str_to_df } iasr_tables = {"maximum_capacity": maximum_capacity} - result = _merge_unit_keyed_properties(generators, iasr_tables, property_map) + exclude_unit_keys = set(["Another_Very_Long_name1"]) + + result = _merge_unit_keyed_properties( + summary, iasr_tables, property_map, exclude_unit_keys + ) expected_result = csv_str_to_df(""" - name, capacity, commissioning_date - BW01, 660.0, 2028-12-01 - EXAMPLE_GEN, 100.0, + name, capacity, commissioning_date + BW01, 660.0, 2028-12-01 + EXAMPLE_GEN, 100.0, + Another_Very_Long_name, 54.0, 2027-05-01 """) - pd.testing.assert_frame_equal(result, expected_result) + pd.testing.assert_frame_equal( + result.sort_values("name").reset_index(drop=True), + expected_result.sort_values("name").reset_index(drop=True), + ) + + +# --- _merge_storage_type_split_properties --- + + +@pytest.mark.parametrize( + "empty_option", + ["full", "empty", "battery_only", "phes_only"], +) +def test_merge_storage_type_split_properties(empty_option, csv_str_to_df): + # Checks the split-merge-concat behaviour across different 'storage' input scenarios + # -> both PHES+Battery, fully empty, battery rows only, phes rows only. + storage = csv_str_to_df(""" + name, power_station, technology + W/HOE#1, Wivenhoe, Hydro + ORANA, Orana BESS, Battery storage (2hrs storage) + """) + empty_options = { + "full": storage.copy(), + "empty": pd.DataFrame(columns=storage.columns), + "battery_only": storage[storage["name"] == "ORANA"].copy(), + "phes_only": storage[storage["name"] == "W/HOE#1"].copy(), + } + iasr_tables = { + "battery_properties": csv_str_to_df(""" + Technology, Charge efficiency_%, Discharge efficiency_% + Battery storage (2hrs storage), 92.0, 92.0 + """), + "pumped_hydro_existing_committed_anticipated_additional_properties": csv_str_to_df(""" + Power Station, Pumping efficiency (%) + Wivenhoe, 64.0 + """), + } + storage_table = empty_options[empty_option] + + result = _merge_storage_type_split_properties(storage_table, iasr_tables) + + expected = csv_str_to_df(""" + name, power_station, technology, efficiency_charge, efficiency_discharge + W/HOE#1, Wivenhoe, Hydro, 80.0, 80.0 + ORANA, Orana BESS, Battery storage (2hrs storage), 92.0, 92.0 + """) + expected_options = { + "full": expected.copy(), + "empty": pd.DataFrame(columns=expected.columns).astype(expected.dtypes), + "battery_only": expected[expected["name"] == "ORANA"].copy(), + "phes_only": expected[expected["name"] == "W/HOE#1"].copy(), + } + pd.testing.assert_frame_equal( + result.sort_values("name").reset_index(drop=True), + expected_options[empty_option].sort_values("name").reset_index(drop=True), + ) # --- _format_commissioning_date --- @@ -274,31 +336,6 @@ def test_merge_minimum_load_bounds_matching_by_technology(csv_str_to_df): pd.testing.assert_frame_equal(result, expected) -def test_merge_minimum_load_empty_generators(csv_str_to_df): - # Nothing to merge onto -- the coal/gas tables aren't even looked at, so an - # empty/invalid one doesn't matter here (mirrors _merge_unit_keyed_properties). - generators = pd.DataFrame(columns=["name", "technology"]) - iasr_tables = { - "coal_minimum_stable_level": pd.DataFrame( - columns=[ - "IASR ID", - "Technology Type", - "Minimum Stable Level (MW)_Typical Lowest Band", - ] - ), - "gpg_min_stable_level_existing_generators": pd.DataFrame( - columns=["IASR ID", "Technology Type", "Min Stable Level (MW)"] - ), - } - - result = _merge_minimum_load(generators, iasr_tables) - - expected = csv_str_to_df(""" - name, technology, minimum_load - """) - pd.testing.assert_frame_equal(result, expected, check_dtype=False) - - @pytest.mark.parametrize( "empty_table", ["coal_minimum_stable_level", "gpg_min_stable_level_existing_generators"], @@ -402,100 +439,64 @@ def test_template_generators_existing_planned(csv_str_to_df): assert len(result) == 2 -def test_template_generators_existing_planned_empty(csv_str_to_df): - columns = [ - "IASR ID / DLT names", - "Power Station", - "Technology Type", - "REZ ID", - "Sub-region", - "Fuel type", - "Fuel cost mapping", - ] - summary = pd.DataFrame(columns=columns) - phes_properties = pd.DataFrame(columns=["Power Station"]) - maximum_capacity = pd.DataFrame( - columns=["IASR ID", "Installed capacity (MW)", "Commissioning date"] - ) - variable_opex = pd.DataFrame( - columns=["IASR ID", "Variable OPEX ($/MWh sent out)1,"] - ) - heat_rates = pd.DataFrame(columns=["IASR ID", "Heat rate (GJ/MWh)"]) - closure_years = pd.DataFrame( - columns=["IASR ID", "Expected Closure Year (Calendar year)"] - ) - coal = pd.DataFrame( - columns=[ - "IASR ID", - "Technology Type", - "Minimum Stable Level (MW)_Typical Lowest Band", - ] - ) - gas = pd.DataFrame(columns=["IASR ID", "Technology Type", "Min Stable Level (MW)"]) - iasr_tables = { - "existing_committed_anticipated_additional_generator_summary": summary, - "pumped_hydro_existing_committed_anticipated_additional_properties": phes_properties, - "maximum_capacity_existing_committed_anticipated_additional_generators": maximum_capacity, - "variable_opex_existing_committed_anticipated_additional_generators": variable_opex, - "heat_rates_existing_committed_anticipated_additional_generators": heat_rates, - "expected_closure_years": closure_years, - "coal_minimum_stable_level": coal, - "gpg_min_stable_level_existing_generators": gas, - } - sub_regional_geography = pd.DataFrame(columns=["geo_id", "geo_type", "region_id"]) - - result = _template_generators_existing_planned( - iasr_tables, "sub_regions", sub_regional_geography - ) - - expected = csv_str_to_df(""" - name, power_station, technology, geo_id, fuel_type, fuel_price_mapping, capacity, vom, heat_rate, commissioning_date, closure_year, minimum_load - """) - pd.testing.assert_frame_equal(result, expected, check_dtype=False) - - # --- _template_storage_existing_planned --- def test_template_storage_existing_planned(csv_str_to_df): + # Wiring only - the behaviour behind each column is covered by the per-helper + # tests above. Columns are compared in order: this is the only test that pins + # _STORAGE_COLUMNS' schema ordering (elsewhere it's compared as a set). summary = csv_str_to_df(""" - IASR ID / DLT names, Power Station, Technology Type - BW01, Bayswater, Steam Sub Critical - WHOE1, Wivenhoe, Hydro - Q8_BATT_2H, Q8 Battery, Battery Storage (2hrs storage) + IASR ID / DLT names, Power Station, Technology Type, REZ ID, Sub-region, Fuel type + W/HOE#1, Wivenhoe, Hydro, Not Applicable, SQ, Water + ORANA, Orana BESS, Battery storage (2hrs storage), N3, CNSW, Battery """) phes_properties = csv_str_to_df(""" - Power Station - Wivenhoe + Power Station, Pumping efficiency (%) + Wivenhoe, 70 + """) + battery_properties = csv_str_to_df(""" + Technology, Energy capacity_Hours, Charge efficiency_%, Discharge efficiency_% + Battery storage (2hrs storage), 2.0, 92.0, 92.0 + """) + maximum_capacity = csv_str_to_df(""" + IASR ID, Installed capacity (MW), Storage Capacity (MWh), Commissioning date + W/HOE#1, 285.0, 3000.0, + ORANA, 415.0, 1660.0, 2026-06-01 + """) + closure_years = csv_str_to_df(""" + IASR ID, Expected Closure Year (Calendar year) + W/HOE#1, 2084 + ORANA, 2066 """) iasr_tables = { "existing_committed_anticipated_additional_generator_summary": summary, "pumped_hydro_existing_committed_anticipated_additional_properties": phes_properties, + "maximum_capacity_existing_committed_anticipated_additional_generators": maximum_capacity, + "expected_closure_years": closure_years, + "battery_properties": battery_properties, } - storage = _template_storage_existing_planned(iasr_tables) - - expected_storage = csv_str_to_df(""" - name, power_station, technology - WHOE1, Wivenhoe, Hydro - Q8_BATT_2H, Q8 Battery, Battery Storage (2hrs storage) + sub_regional_geography = csv_str_to_df(""" + geo_id, geo_type, region_id + SQ, subregion, QLD + N3, rez, NSW """) - pd.testing.assert_frame_equal(storage.reset_index(drop=True), expected_storage) - -def test_template_storage_existing_planned_empty(csv_str_to_df): - summary = pd.DataFrame( - columns=["IASR ID / DLT names", "Power Station", "Technology Type"] + result = _template_storage_existing_planned( + iasr_tables, "sub_regions", sub_regional_geography ) - phes_properties = pd.DataFrame(columns=["Power Station"]) - iasr_tables = { - "existing_committed_anticipated_additional_generator_summary": summary, - "pumped_hydro_existing_committed_anticipated_additional_properties": phes_properties, - } - - storage = _template_storage_existing_planned(iasr_tables) - - expected_storage = csv_str_to_df(""" - name, power_station, technology - """) - pd.testing.assert_frame_equal(storage, expected_storage, check_dtype=False) + assert list(result.columns) == [ + "name", + "power_station", + "technology", + "geo_id", + "fuel_type", + "capacity", + "storage_capacity", + "efficiency_charge", + "efficiency_discharge", + "commissioning_date", + "closure_year", + ] + assert len(result) == 2 diff --git a/tests/test_templater/test_helpers.py b/tests/test_templater/test_helpers.py index feebcb99..03e9462d 100644 --- a/tests/test_templater/test_helpers.py +++ b/tests/test_templater/test_helpers.py @@ -1,8 +1,11 @@ +import logging +import re + import pandas as pd import pytest from ispypsa.templater.helpers import ( - _apply_known_value_replacement, + _apply_iasr_table_replacements, _assert_table_valid, _build_geo_region_lookup, _derive_phes_symmetric_efficiency, @@ -15,6 +18,7 @@ _looks_like_financial_year, _manual_remove_footnotes_from_generator_names, _map_geo_id_to_granularity, + _merge_category_keyed_properties, _pick_location, _required_property_columns, _rez_name_to_id_mapping, @@ -478,8 +482,21 @@ def test_derive_phes_symmetric_efficiency(csv_str_to_df): result = _derive_phes_symmetric_efficiency(phes) expected = csv_str_to_df(""" - name, round_trip_efficiency, efficiency_charge, efficiency_discharge - NQ Pumped Hydro - 24h, 81.0, 90.0, 90.0 + name, efficiency_charge, efficiency_discharge + NQ Pumped Hydro - 24h, 90.0, 90.0 + """) + pd.testing.assert_frame_equal(result, expected, check_exact=False, rtol=1e-6) + + +def test_derive_phes_symmetric_efficiency_empty(csv_str_to_df): + phes = csv_str_to_df(""" + name, round_trip_efficiency + """) + + result = _derive_phes_symmetric_efficiency(phes) + + expected = csv_str_to_df(""" + name, efficiency_charge, efficiency_discharge """) pd.testing.assert_frame_equal(result, expected, check_exact=False, rtol=1e-6) @@ -616,42 +633,102 @@ def test_is_subregion_geo_id(csv_str_to_df): pd.testing.assert_series_equal(result, expected) -# --- _apply_known_value_replacement --- +# --- _apply_iasr_table_replacements --- -def test_apply_known_value_replacement(csv_str_to_df): +def test_apply_iasr_table_replacements(csv_str_to_df): maximum_capacity = csv_str_to_df(""" IASR ID, Installed capacity (MW) KiataWF1, 30.0 BW01, 660.0 """) + phes_properties = csv_str_to_df(""" + Power Station, Installed capacity (MW), Storage capacity (hours), Pumping efficiency (%) + Wivenhoe, 570, 10.0, 70 + Shoalhaven, 240, 64.0, 70 + Borumba, 1998, 24.0, 76 + """) iasr_tables = { "maximum_capacity": maximum_capacity, + "phes_properties": phes_properties, "some_other_table": pd.DataFrame({"col": [1]}), } - correction = dict( - table_name="maximum_capacity", - column="IASR ID", - replacements={"KiataWF1": "KIATAWF1"}, - ) - result = _apply_known_value_replacement(iasr_tables, correction) + corrections = [ + dict( + table_name="maximum_capacity", + column="IASR ID", + replacements={"KiataWF1": "KIATAWF1"}, + ), + dict( + table_name="phes_properties", + column="Power Station", + replacements={"Borumba": "QEJP - Borumba"}, + ), + ] + result = _apply_iasr_table_replacements(iasr_tables, corrections) - expected = csv_str_to_df(""" + expected_capacity = csv_str_to_df(""" IASR ID, Installed capacity (MW) KIATAWF1, 30.0 BW01, 660.0 """) - pd.testing.assert_frame_equal(result["maximum_capacity"], expected) + expected_phes_props = csv_str_to_df(""" + Power Station, Installed capacity (MW), Storage capacity (hours), Pumping efficiency (%) + Wivenhoe, 570, 10.0, 70 + Shoalhaven, 240, 64.0, 70 + QEJP - Borumba, 1998, 24.0, 76 + """) + pd.testing.assert_frame_equal(result["maximum_capacity"], expected_capacity) + pd.testing.assert_frame_equal(result["phes_properties"], expected_phes_props) # Other tables pass through untouched; input dict itself isn't mutated. assert result["some_other_table"] is iasr_tables["some_other_table"] - unmutated = csv_str_to_df(""" + + unmutated_capacity = csv_str_to_df(""" IASR ID, Installed capacity (MW) KiataWF1, 30.0 BW01, 660.0 """) - pd.testing.assert_frame_equal(iasr_tables["maximum_capacity"], unmutated) + unmutated_phes_props = csv_str_to_df(""" + Power Station, Installed capacity (MW), Storage capacity (hours), Pumping efficiency (%) + Wivenhoe, 570, 10.0, 70 + Shoalhaven, 240, 64.0, 70 + Borumba, 1998, 24.0, 76 + """) + pd.testing.assert_frame_equal(iasr_tables["maximum_capacity"], unmutated_capacity) + pd.testing.assert_frame_equal(iasr_tables["phes_properties"], unmutated_phes_props) + + +def test_apply_iasr_table_replacements_missing_expected_col(csv_str_to_df): + # Checks that replacement only applies to specified columns, and no error if a + # named column is missing. Column presence is asserted later by _assert_table_valid + maximum_capacity = csv_str_to_df(""" + unit_name, Installed capacity (MW) + KiataWF1, 30.0 + BW01, 660.0 + """) + iasr_tables = { + "maximum_capacity": maximum_capacity, + } + corrections = [ + dict( + table_name="maximum_capacity", + column="IASR ID", + replacements={"KiataWF1": "KIATAWF1"}, + ), + ] + result = _apply_iasr_table_replacements(iasr_tables, corrections) + + expected_unchanged = csv_str_to_df(""" + unit_name, Installed capacity (MW) + KiataWF1, 30.0 + BW01, 660.0 + """) + pd.testing.assert_frame_equal( + result["maximum_capacity"].sort_values("unit_name").reset_index(drop=True), + expected_unchanged.sort_values("unit_name").reset_index(drop=True), + ) # --- _group_properties_by_source --- @@ -769,6 +846,126 @@ def test_get_property_value_map_raises_on_typo(csv_str_to_df): _get_property_value_map(table, attrs) +# --- _merge_category_keyed_properties --- + + +def test_merge_category_keyed_properties(csv_str_to_df): + # SUPER GENERIC check that 'df_key_col' input correctly sets the column-to-map + # onto the input dataframe. + df = csv_str_to_df(""" + first_col, second_col, third_col + Big Apples, Small Oranges, Pink Bananas + Pink Bananas, Big Apples, Small Oranges + Small Oranges, Pink Bananas, Big Apples + Big Apples, Small Oranges, Pink Bananas + """) + tables = { + "fruit_costs": csv_str_to_df(""" + Fruits, Costs + Big Apples, 10.0 + Small 0ranges, 15.0 + Pink Bananas, 50.0 + """), # 'Small 0ranges' <-> 'Small Oranges' via key resolution + } + + property_map = { + "second_col_costs": dict( + table="fruit_costs", + key_col="Fruits", + value_col="Costs", + ), + } + + result = _merge_category_keyed_properties(df, tables, property_map, "second_col") + + expected = csv_str_to_df(""" + first_col, second_col, third_col, second_col_costs + Big Apples, Small Oranges, Pink Bananas, 15.0 + Pink Bananas, Big Apples, Small Oranges, 10.0 + Small Oranges, Pink Bananas, Big Apples, 50.0 + Big Apples, Small Oranges, Pink Bananas, 15.0 + """) + + pd.testing.assert_frame_equal(result, expected) + + +def test_merge_category_keyed_properties_shared_source_merged_once( + csv_str_to_df, caplog +): + # storage_hours and efficiency_charge both come from battery_properties/Technology + # (as in _STORAGE_BATTERY_PROPERTY_MAP): both are merged correctly in one pass, + # NaN property values are retained untouched, and - because they share a source + # table - the fuzzy match against it runs once, so a corrected technology name is + # logged once, not once per property sourced from that table. + new_entrants = csv_str_to_df(""" + name, technology + NQ Battery - 2h, battery storage (2hrs storage) + NQ CCGT, CCGT + """) + property_map = { + "storage_hours": { + "table": "battery_properties", + "key_col": "Technology", + "value_col": "Energy capacity_Hours", + }, + "efficiency_charge": { + "table": "battery_properties", + "key_col": "Technology", + "value_col": "Charge efficiency_%", + }, + } + iasr_tables = { + "battery_properties": csv_str_to_df(""" + Technology, Energy capacity_Hours, Charge efficiency_% + Battery Storage (2hrs storage), 2.0, 92.0 + CCGT, , + """), + } + + with caplog.at_level("INFO"): + result = _merge_category_keyed_properties( + new_entrants, iasr_tables, property_map, "technology" + ) + + expected = csv_str_to_df(""" + name, technology, storage_hours, efficiency_charge + NQ Battery - 2h, battery storage (2hrs storage), 2.0, 92.0 + NQ CCGT, CCGT, , + """) + pd.testing.assert_frame_equal(result, expected) + + msg = ( + "'battery storage (2hrs storage)' matched to " + "'Battery Storage (2hrs storage)' whilst merging properties " + "from 'battery_properties'" + ) + assert caplog.messages.count(msg) == 1 + + +def test_merge_category_keyed_properties_raises_on_invalid_source_table(csv_str_to_df): + # Regression: confirms the source table is actually validated before merging. + # Exact raise behaviour is covered by _assert_table_valid's own tests. + new_entrants = csv_str_to_df(""" + name, technology + SQ CCGT, CCGT + """) + property_map = { + "fom": { + "table": "fixed_opex_new_entrants", + "key_col": "Technology", + "value_col": "Base value", + } + } + iasr_tables = { + "fixed_opex_new_entrants": pd.DataFrame(columns=["Technology", "Base value"]), + } + + with pytest.raises(ValueError): + _merge_category_keyed_properties( + new_entrants, iasr_tables, property_map, "technology" + ) + + # --- _assert_table_valid --- diff --git a/tests/test_templater/test_new_entrants.py b/tests/test_templater/test_new_entrants.py index 161df3a6..0187448d 100644 --- a/tests/test_templater/test_new_entrants.py +++ b/tests/test_templater/test_new_entrants.py @@ -13,7 +13,6 @@ _merge_lcf_build, _merge_lcf_om, _merge_phes_properties, - _merge_properties, _override_botn_technology, _rekey_names_to_collapsed_geo_id, _reshape_technology_specific_lcfs, @@ -167,83 +166,6 @@ def test_template_storage_new_entrant(csv_str_to_df): assert len(result) == 3 -# --- _merge_properties --- -# (_group_properties_by_source, _required_property_columns and -# _get_property_value_map are shared with existing_planned.py and now live in, and -# are tested in, helpers.py / test_helpers.py.) - - -def test_merge_properties(csv_str_to_df, caplog): - # storage_hours and efficiency_charge both come from battery_properties/Technology - # (as in _STORAGE_BATTERY_PROPERTY_MAP): both are merged correctly in one pass, - # NaN property values are retained untouched, and - because they share a source - # table - the fuzzy match against it runs once, so a corrected technology name is - # logged once, not once per property sourced from that table. - new_entrants = csv_str_to_df(""" - name, technology - NQ Battery - 2h, battery storage (2hrs storage) - NQ CCGT, CCGT - """) - property_map = { - "storage_hours": { - "table": "battery_properties", - "key_col": "Technology", - "value_col": "Energy capacity_Hours", - }, - "efficiency_charge": { - "table": "battery_properties", - "key_col": "Technology", - "value_col": "Charge efficiency_%", - }, - } - iasr_tables = { - "battery_properties": csv_str_to_df(""" - Technology, Energy capacity_Hours, Charge efficiency_% - Battery Storage (2hrs storage), 2.0, 92.0 - CCGT, , - """), - } - - with caplog.at_level("INFO"): - result = _merge_properties(new_entrants, iasr_tables, property_map) - - expected = csv_str_to_df(""" - name, technology, storage_hours, efficiency_charge - NQ Battery - 2h, battery storage (2hrs storage), 2.0, 92.0 - NQ CCGT, CCGT, , - """) - pd.testing.assert_frame_equal(result, expected) - - msg = ( - "'battery storage (2hrs storage)' matched to " - "'Battery Storage (2hrs storage)' whilst merging new entrant properties " - "from 'battery_properties'" - ) - assert caplog.messages.count(msg) == 1 - - -def test_merge_properties_raises_on_invalid_source_table(csv_str_to_df): - # Regression: confirms the source table is actually validated before merging. - # Exact raise behaviour is covered by _assert_table_valid's own tests. - new_entrants = csv_str_to_df(""" - name, technology - SQ CCGT, CCGT - """) - property_map = { - "fom": { - "table": "fixed_opex_new_entrants", - "key_col": "Technology", - "value_col": "Base value", - } - } - iasr_tables = { - "fixed_opex_new_entrants": pd.DataFrame(columns=["Technology", "Base value"]), - } - - with pytest.raises(ValueError): - _merge_properties(new_entrants, iasr_tables, property_map) - - # --- _merge_phes_properties / _override_botn_technology / _derive_phes_symmetric_efficiency --- @@ -267,9 +189,9 @@ def test_merge_phes_properties(csv_str_to_df): result = _merge_phes_properties(phes, iasr_tables) expected = csv_str_to_df(""" - name, technology, storage_hours, round_trip_efficiency, efficiency_charge, efficiency_discharge - NQ Pumped Hydro - 24h, Pumped Hydro (24hrs storage), 24.0, 64.0, 80.0, 80.0 - BOTN - Cethana - 20h, BOTN - Cethana, 20.0, 81.0, 90.0, 90.0 + name, technology, storage_hours, efficiency_charge, efficiency_discharge + NQ Pumped Hydro - 24h, Pumped Hydro (24hrs storage), 24.0, 80.0, 80.0 + BOTN - Cethana - 20h, BOTN - Cethana, 20.0, 90.0, 90.0 """) pd.testing.assert_frame_equal(result, expected, check_exact=False, rtol=1e-6) @@ -334,12 +256,12 @@ def test_merge_phes_properties_empty(csv_str_to_df): result = _merge_phes_properties(phes, iasr_tables) expected = csv_str_to_df(""" - name, technology, storage_hours, round_trip_efficiency, efficiency_charge, efficiency_discharge + name, technology, storage_hours, efficiency_charge, efficiency_discharge """) pd.testing.assert_frame_equal(result, expected, check_dtype=False) -# (BOTN's pumped-hydro key correction is now a plain _apply_known_value_replacement +# (BOTN's pumped-hydro key correction is now a plain _apply_iasr_table_replacements # call -- see helpers.py / test_helpers.py for that mechanism's own tests. Its wiring # here is covered incidentally by test_merge_phes_properties above, which already # exercises a full-spelling BOTN row resolving correctly.)