From 0a9569e502b3ebef31af01593eda8ef7b95be48b Mon Sep 17 00:00:00 2001 From: EllieKallmier <61219730+EllieKallmier@users.noreply.github.com> Date: Thu, 3 Sep 2026 16:56:26 +1000 Subject: [PATCH] fix: new entrant renaming after granularity collapse - only re-key old geo_id in name --- src/ispypsa/templater/new_entrants.py | 81 ++++++++++++++--------- tests/test_templater/test_new_entrants.py | 36 +++++++++- 2 files changed, 85 insertions(+), 32 deletions(-) diff --git a/src/ispypsa/templater/new_entrants.py b/src/ispypsa/templater/new_entrants.py index 90d41721..fc00bb45 100644 --- a/src/ispypsa/templater/new_entrants.py +++ b/src/ispypsa/templater/new_entrants.py @@ -152,6 +152,10 @@ ("NSA", "CSA"), # (geo_id, Regional build cost zone) } +# Allowed 'extra' subregion (not in sub_regional_geography) present in the names of +# some new entrant gas plant; see Open-ISP/ISPyPSA#131 +_EXTRA_SUBREGION_IN_NAMES = {"WOO"} + # Data scale diff for 'BOTN - Cethana' in LCF table: see first comment on Open-ISP/ISPyPSA#131. _LCF_COLUMNS_IN_PERCENT = ["BOTN - Cethana"] @@ -342,8 +346,9 @@ def _collapse_geo_id_to_granularity( 1. Splits ``new_entrants`` into REZ rows (left untouched) and subregion rows. 2. Subregion rows get grouped by ``group_key_columns`` + the re-keyed geo_id and averaged over ``value_columns``. - 3. Aggregated rows' 'name' set to "{geo_id} {technology}" (except BOTN - see - ``_name_collapsed_rows``). + 3. Aggregated rows keep the first 'name' picked by the groupby, with any stale + leading sub-region-style token (a real geo_id, or a known extra like "WOO") + replaced by the new, collapsed geo_id — see ``_rekey_names_to_collapsed_geo_id``. 4. Returns concatted REZ rows and aggregated rows. Args: @@ -368,8 +373,8 @@ def _collapse_geo_id_to_granularity( SNW subregion NSW returns: - name technology geo_id lcf_build - NSW OCGT (small GT) OCGT (small GT) NSW 102.0 # mean(104, 100) + name technology geo_id lcf_build + NSW OCGT Small OCGT (small GT) NSW 102.0 # mean(104, 100) """ if regional_granularity == "sub_regions": return new_entrants @@ -380,11 +385,17 @@ def _collapse_geo_id_to_granularity( if to_collapse.empty: return new_entrants - to_collapse["geo_id"] = _map_geo_id_to_granularity( - to_collapse["geo_id"], regional_granularity, sub_regional_geography + old_geo_ids = to_collapse["geo_id"].copy() + new_geo_ids = _map_geo_id_to_granularity( + old_geo_ids, regional_granularity, sub_regional_geography + ) + collapsed = _aggregate_by_geo_id( + to_collapse.assign(geo_id=new_geo_ids), group_key_columns, value_columns + ) + collapsed["name"] = _rekey_names_to_collapsed_geo_id( + collapsed, + set(old_geo_ids) | _EXTRA_SUBREGION_IN_NAMES, ) - collapsed = _aggregate_by_geo_id(to_collapse, group_key_columns, value_columns) - collapsed = _name_collapsed_rows(collapsed) return pd.concat([unchanged, collapsed], ignore_index=True)[new_entrants.columns] @@ -394,38 +405,48 @@ def _aggregate_by_geo_id( group_key_columns: list[str], value_columns: list[str], ) -> pd.DataFrame: - """Groups by ``group_key_columns`` + 'geo_id' and averages ``value_columns``.""" - # 'dropna=False' set to keep thermal generator rows (w/ NaN 'resource_type') + """Groups by ``group_key_columns`` + 'geo_id', averages ``value_columns``, keeps + the first instance of 'name' for each group.""" return new_entrants.groupby( group_key_columns + ["geo_id"], dropna=False, as_index=False - )[value_columns].mean() + ).agg({"name": "first", **{col: "mean" for col in value_columns}}) -# TODO (coming in next PR): fix this to keep naming convention from AEMO sources even when -# granularity collapses. NOTE: IASR names have slightly different order/ -# convention than trace names, but for VRE (REZ-based) so less important here. -def _name_collapsed_rows(collapsed: pd.DataFrame) -> pd.DataFrame: - """Sets 'name' on merged rows to "{geo_id} {technology}". +def _rekey_names_to_collapsed_geo_id( + new_entrants: pd.DataFrame, known_name_prefixes: set[str] +) -> pd.Series: + """Replaces a stale leading sub-region-style token in each name with the row's + (post-collapse) geo_id. - The lone documented exception is BOTN - Cethana (see ``_BOTN_CETHANA_DETAILS``): - a named, site-specific project rather than a generic technology archetype, which - keeps its original 'name'. + Runs on the already-aggregated frame, after ``.groupby(...).first()`` has picked + one 'name' per group — so it doesn't matter which row's name survived the pick; + every stale geo_id-like prefix gets normalised to the same, correct new geo_id. I/O Example: - collapsed: - technology geo_id - OCGT (small GT) NSW - BOTN - Cethana TAS + df: + name technology geo_id ... + NQ OCGT Small OCGT (small GT) NEM ... + WOO OCGT Large OCGT (large GT) NEM ... + BOTN - Cethana - 20h BOTN - Cethana NEM ... + + known_name_prefixes: {"NQ", "WOO", "SNW", ...} # real sub_regions + "WOO" returns: - technology geo_id name - OCGT (small GT) NSW NSW OCGT (small GT) - BOTN - Cethana TAS BOTN - Cethana - 20h # original name kept + name + NEM OCGT Small + NEM OCGT Large + BOTN - Cethana - 20h + + # BOTN doesn't start with any known prefix, so it's untouched - no special case needed. """ - fresh_name = collapsed["geo_id"] + " " + collapsed["technology"] - is_botn = collapsed["technology"] == _BOTN_CETHANA_DETAILS["name"] - collapsed["name"] = fresh_name.mask(is_botn, _BOTN_CETHANA_DETAILS["full_name"]) - return collapsed + + parts = new_entrants["name"].str.partition() + parts.columns = ["old_prefix", "separator", "rest_of_name"] + + rekeyed_names = new_entrants["geo_id"].str.cat(parts[["separator", "rest_of_name"]]) + is_known_prefix = parts["old_prefix"].isin(known_name_prefixes) + + return new_entrants["name"].where(~is_known_prefix, rekeyed_names) # --- locational cost factor (LCF) helpers --- diff --git a/tests/test_templater/test_new_entrants.py b/tests/test_templater/test_new_entrants.py index 133cb46e..22b4aaa1 100644 --- a/tests/test_templater/test_new_entrants.py +++ b/tests/test_templater/test_new_entrants.py @@ -15,6 +15,7 @@ _merge_phes_properties, _merge_properties, _override_botn_technology, + _rekey_names_to_collapsed_geo_id, _reshape_technology_specific_lcfs, _template_generators_new_entrant, _template_storage_new_entrant, @@ -406,7 +407,7 @@ def test_collapse_geo_id_to_granularity_averages_across_sub_regions(csv_str_to_d expected = csv_str_to_df(""" name, technology, geo_id, value Q1_WH, Wind, Q1, 999.0 - NSW OCGT, OCGT, NSW, 102.0 + NSW OCGT Small, OCGT, NSW, 102.0 BOTN - Cethana - 20h, BOTN - Cethana, TAS, 100.0 """) pd.testing.assert_frame_equal( @@ -440,7 +441,7 @@ def test_collapse_geo_id_to_granularity_single_region_maps_to_nem(csv_str_to_df) expected = csv_str_to_df(""" name, technology, geo_id, value - NEM OCGT, OCGT, NEM, 106.0 + NEM OCGT Small, OCGT, NEM, 106.0 BOTN - Cethana - 20h, BOTN - Cethana, NEM, 100.0 """) pd.testing.assert_frame_equal( @@ -473,6 +474,37 @@ def test_collapse_geo_id_to_granularity_empty_input(csv_str_to_df): pd.testing.assert_frame_equal(result, expected) +def test_rekey_names_to_collapsed_geo_id(csv_str_to_df): + # Specifically checks that the 'same prefix, different geo_id' case correctly + # re-keys based on the corresponding row-specific geo_id value, and that + # 'unknown' (not 'old' geo_id values) prefixed names pass through unchanged. + + # only necessary columns - abbreviated input + new_entrants = csv_str_to_df(""" + name, geo_id + NSA Pumped Hydro - 10h, SA + WOO CCGT, NSW + NQ OCGT Small, QLD + NQ Biomass, VIC + Named Generator, TAS + """) + + known_prefixes = {"NSA", "WOO", "NQ"} + result = _rekey_names_to_collapsed_geo_id(new_entrants, known_prefixes) + + expected = pd.Series( + [ + "SA Pumped Hydro - 10h", + "NSW CCGT", + "QLD OCGT Small", + "VIC Biomass", + "Named Generator", + ], + name="name", + ) + pd.testing.assert_series_equal(result, expected) + + # --- _add_resource_type (generator-specific) ---