From 5d21fff387450919da4c1612f7e352e3fb64e18f Mon Sep 17 00:00:00 2001 From: Ben Elliston Date: Fri, 11 Sep 2026 20:07:11 +1000 Subject: [PATCH] Fix D-class Ruff messages about docstring formatting. --- .../construct_reference_year_mapping.py | 4 +- src/isp_trace_parser/demand_traces.py | 13 +-- src/isp_trace_parser/get_data.py | 94 ++++++++----------- src/isp_trace_parser/mappings/__init__.py | 1 + src/isp_trace_parser/optimise_parquet.py | 3 +- src/isp_trace_parser/remote/download.py | 4 +- .../resource_trace_metadata.py | 1 - src/isp_trace_parser/solar_traces.py | 23 +++-- src/isp_trace_parser/trace_formatter.py | 6 +- .../trace_restructure_helper_functions.py | 5 +- src/isp_trace_parser/wind_traces.py | 21 ++--- 11 files changed, 75 insertions(+), 100 deletions(-) diff --git a/src/isp_trace_parser/construct_reference_year_mapping.py b/src/isp_trace_parser/construct_reference_year_mapping.py index b79ba76..2a4e0fe 100644 --- a/src/isp_trace_parser/construct_reference_year_mapping.py +++ b/src/isp_trace_parser/construct_reference_year_mapping.py @@ -14,10 +14,9 @@ def construct_reference_year_mapping( start_year: int, end_year: int, reference_years: list[int] ) -> dict: - """Constructs a dictionary mapping a sequence of modeling years to a cycle of reference years. + """Construct a dictionary mapping a sequence of modeling years to a cycle of reference years. Examples: - >>> construct_reference_year_mapping( ... start_year=2030, ... end_year=2035, @@ -29,6 +28,7 @@ def construct_reference_year_mapping( start_year: int, first year in sequence of modelling years end_year: int, last year in sequence of modelling years reference_years: list[int], list of reference years to cycle through when constructing the mapping. + """ input_validation.start_year_before_end_year(start_year, end_year) years = range(start_year, end_year + 1) diff --git a/src/isp_trace_parser/demand_traces.py b/src/isp_trace_parser/demand_traces.py index 4a0a690..898fabf 100644 --- a/src/isp_trace_parser/demand_traces.py +++ b/src/isp_trace_parser/demand_traces.py @@ -30,7 +30,6 @@ class DemandMetadataFilter(BaseModel): included then only traces with metadata matching the values in the corresponding list will be parsed. Examples: - Filter for only subregions that are in a list of names. >>> metadata_filters = DemandMetadataFilter( @@ -50,6 +49,7 @@ class DemandMetadataFilter(BaseModel): poe: list of POE levels, only including "POE10" and "POE50" demand_type, list of demand types, only including "OPSO_MODELLING", "OPSO_MODELLING_PVLITE", and "PV_TOT" reference_year: list of ints specifying reference_years + """ subregion: list[str] | None = None @@ -71,7 +71,7 @@ def parse_demand_traces( use_concurrency: bool = True, filters: DemandMetadataFilter | None = None, ) -> None: - """Takes a directory with AEMO demand trace data and reformats the data, saving it to a new directory. + """Take a directory with AEMO demand trace data and reformats the data, saving it to a new directory. AEMO demand trace data comes in CSVs with columns specifying the year, day, and month, and data columns (labeled 01, 02, ... 48) storing the demand values for each half hour of the day. The file name of the CSV @@ -94,7 +94,6 @@ def parse_demand_traces( metadata value present in the corresponding filter list will be passed, see examples below. Examples: - Parse whole directory of trace data. >>> parse_demand_traces( @@ -134,6 +133,7 @@ def parse_demand_traces( attribute is not set, no filtering on that attribute occurs. See example. Returns: None + """ input_directory = input_validation.input_directory(input_directory) parsed_directory = input_validation.parsed_directory(parsed_directory) @@ -171,8 +171,7 @@ def restructure_demand_file( output_directory: Path, filters: DemandMetadataFilter | None = None, ) -> None: - """ - Restructures a single demand trace file and saves it as parquet. + """Restructures a single demand trace file and saves it as parquet. The output filename is the AEMO input filename with the .csv suffix replaced by .parquet (e.g. CNSW_RefYear_2011_HYDROGEN_EXPORT_POE10_OPSO_MODELLING.csv @@ -212,6 +211,7 @@ def restructure_demand_file( ... ) # doctest: +SKIP # This will process the input file and save it in parquet format in the specified output directory + """ file_metadata["scenario"] = demand_scenario_mapping[file_metadata["scenario"]] @@ -226,11 +226,12 @@ def restructure_demand_file( def _frame_with_metadata(trace: pl.DataFrame, file_metadata: dict) -> pl.DataFrame: - """Adds metadata fields as columns to a trace DataFrame. + """Add metadata fields as columns to a trace DataFrame. Args: trace: The trace data Polars Dataframe to add data to. file_metadata: Dict containing metadata (subregion, reference_year, scenario, poe, and demand_type) + """ return trace.with_columns( subregion=pl.lit(file_metadata["subregion"]), diff --git a/src/isp_trace_parser/get_data.py b/src/isp_trace_parser/get_data.py index 1cf4f91..3163be8 100644 --- a/src/isp_trace_parser/get_data.py +++ b/src/isp_trace_parser/get_data.py @@ -17,8 +17,7 @@ def _year_range_to_dt_range( start_year: int, end_year: int, year_type: Literal["fy", "calendar"] = "fy" ) -> datetime.datetime: - """ - Convert year range to datetime boundaries for efficient time filtering. + """Convert year range to datetime boundaries for efficient time filtering. Handles both financial year (FY) and calendar year conventions. For FY, uses year-ending nomenclature where FY2022 spans July 1, 2021 to July 1, 2022. @@ -37,8 +36,8 @@ def _year_range_to_dt_range( >>> _year_range_to_dt_range(2022, 2024, year_type="calendar") (datetime.datetime(2022, 1, 1, 0, 0), datetime.datetime(2025, 1, 1, 0, 0)) - """ + """ if year_type == "fy": return datetime.datetime(start_year - 1, 7, 1), datetime.datetime( end_year, 7, 1 @@ -60,8 +59,7 @@ def _query_parquet_single_reference_year( select_columns: list[str] | None = None, year_type: Literal["fy", "calendar"] = "fy", ) -> pd.DataFrame: - """ - Generic function to query parquet files with flexible column filters. + """Query parquet files with flexible column filters. Args: start_year: Start of time window @@ -78,6 +76,7 @@ def _query_parquet_single_reference_year( Returns: pd.DataFrame with selected columns, sorted by datetime + """ start_dt, end_dt = _year_range_to_dt_range(start_year, end_year, year_type) @@ -124,8 +123,7 @@ def _query_parquet_single_reference_year( def _query_parquet_multiple_reference_years( reference_year_mapping: dict[int, int], **kwargs: any ) -> pd.DataFrame: - """ - Query parquet files across multiple reference years. + """Query parquet files across multiple reference years. Iteratively calls _query_parquet_single_reference_year for each year-reference_year pair and concatenates the results. @@ -136,6 +134,7 @@ def _query_parquet_multiple_reference_years( Returns: pd.DataFrame with concatenated results from all years + """ data = [] for year, reference_year in reference_year_mapping.items(): @@ -157,8 +156,7 @@ def get_project_single_reference_year( year_type: Literal["fy", "calendar"] = "fy", select_columns: list[str] | None = None, ) -> pd.DataFrame: - """ - Query project trace data for a single reference year. + """Query project trace data for a single reference year. Retrieves trace data for one or more projects within a specified time window. When querying multiple projects (as a list), the 'project' column is automatically @@ -227,6 +225,7 @@ def get_project_single_reference_year( 70175 2024-07-01 00:00:00 0.076900 Bango 973 Wind Farm [70176 rows x 3 columns] + """ return _query_parquet_single_reference_year( start_year=start_year, @@ -250,8 +249,7 @@ def get_zone_single_reference_year( year_type: Literal["fy", "calendar"] = "fy", select_columns: list[str] | None = None, ) -> pd.DataFrame: - """ - Query zone trace data for a single reference year. + """Query zone trace data for a single reference year. Retrieves trace data for one or more zones and resource types within a specified time window. When querying multiple zones (as a list), the 'zone' column is @@ -323,6 +321,7 @@ def get_zone_single_reference_year( 105263 2024-07-01 00:00:00 0.0 N1 [105264 rows x 3 columns] + """ return _query_parquet_single_reference_year( start_year=start_year, @@ -348,8 +347,7 @@ def get_demand_single_reference_year( year_type: Literal["fy", "calendar"] = "fy", select_columns: list[str] | None = None, ) -> pd.DataFrame: - """ - Query demand trace data for a single reference year. + """Query demand trace data for a single reference year. Retrieves demand trace data for specified scenario, subregion, demand type, and probability of exceedance (POE) within a time window. When querying with multiple @@ -428,6 +426,7 @@ def get_demand_single_reference_year( 140351 2025-07-01 00:00:00 1952.508153 OPSO_MODELLING CSA [140352 rows x 4 columns] + """ return _query_parquet_single_reference_year( start_year=start_year, @@ -453,8 +452,7 @@ def get_project_multiple_reference_years( year_type: Literal["fy", "calendar"] = "fy", select_columns: list[str] | None = None, ) -> pd.DataFrame: - """ - Query project trace data across multiple reference years. + """Query project trace data across multiple reference years. Retrieves trace data for one or more projects across different years, each potentially using a different reference year. Results from all years are @@ -524,6 +522,7 @@ def get_project_multiple_reference_years( 70175 2025-07-01 00:00:00 0.037577 Bango 973 Wind Farm 2012 [70176 rows x 4 columns] + """ return _query_parquet_multiple_reference_years( reference_year_mapping=reference_year_mapping, @@ -543,8 +542,7 @@ def get_zone_multiple_reference_years( year_type: Literal["fy", "calendar"] = "fy", select_columns: list[str] | None = None, ) -> pd.DataFrame: - """ - Query zone trace data across multiple reference years. + """Query zone trace data across multiple reference years. Retrieves trace data for one or more zones and resource types across different years, each potentially using a different reference year. Results from all years @@ -617,6 +615,7 @@ def get_zone_multiple_reference_years( 105263 2025-07-01 00:00:00 0.0 N1 2012 [105264 rows x 4 columns] + """ return _query_parquet_multiple_reference_years( reference_year_mapping=reference_year_mapping, @@ -638,8 +637,7 @@ def get_demand_multiple_reference_years( year_type: Literal["fy", "calendar"] = "fy", select_columns: list[str] | None = None, ) -> pd.DataFrame: - """ - Query demand trace data across multiple reference years. + """Query demand trace data across multiple reference years. Retrieves demand trace data for specified scenario, subregion, demand type, and probability of exceedance (POE) across different years, each potentially using a @@ -719,6 +717,7 @@ def get_demand_multiple_reference_years( 70175 2025-07-01 00:00:00 5659.380906 VIC 2012 [70176 rows x 4 columns] + """ return _query_parquet_multiple_reference_years( reference_year_mapping=reference_year_mapping, @@ -752,13 +751,11 @@ def solar_project_single_reference_year( directory: str | Path, year_type: Literal["fy", "calendar"] = "fy", ) -> pd.DataFrame: - """ - Pass-through function to keep backwards capability with previos API + """Pass-through function to keep backwards capability with previos API. Reads solar project trace data from an output directory created by isp_trace_parser.solar_trace_parser. Examples: - >>> solar_project_single_reference_year( ... start_year=2022, ... end_year=2024, @@ -793,8 +790,8 @@ def solar_project_single_reference_year( FY2015/2016). If 'calendar', then filtering is by calendar year. Returns: pd.DataFrame with columns datetime and value - """ + """ return get_project_single_reference_year( start_year=start_year, end_year=end_year, @@ -814,11 +811,9 @@ def wind_project_single_reference_year( directory: str | Path, year_type: Literal["fy", "calendar"] = "fy", ) -> pd.DataFrame: - """ - Pass-through function to keep backwards capability with previos API + """Pass-through function to keep backwards capability with previos API. Examples: - >>> wind_project_single_reference_year( ... start_year=2022, ... end_year=2024, @@ -854,6 +849,7 @@ def wind_project_single_reference_year( FY2015/2016). If 'calendar', then filtering is by calendar year. Returns: pd.DataFrame with columns datetime and value + """ return get_project_single_reference_year( start_year=start_year, @@ -872,14 +868,12 @@ def solar_project_multiple_reference_years( directory: str | Path, year_type: Literal["fy", "calendar"] = "fy", ) -> pd.DataFrame: - """ - Pass-through function to keep backwards capability with previos API + """Pass-through function to keep backwards capability with previos API. Reads solar project trace data from an output directory created by isp_trace_parser.solar_trace_parser. Examples: - >>> solar_project_multiple_reference_years( ... reference_years={2022: 2011, 2024: 2012}, ... project='Adelaide Desalination Plant Solar Farm', @@ -912,6 +906,7 @@ def solar_project_multiple_reference_years( FY2015/2016). If 'calendar', then filtering is by calendar year. Returns: pd.DataFrame with columns datetime and value + """ return get_project_multiple_reference_years( reference_year_mapping=reference_years, @@ -930,13 +925,11 @@ def solar_area_single_reference_year( directory: str | Path, year_type: Literal["fy", "calendar"] = "fy", ) -> pd.DataFrame: - """ - Pass-through function to keep backwards capability with previos API + """Pass-through function to keep backwards capability with previos API. Reads solar area trace data from an output directory created by isp_trace_parser.solar_trace_parser. Examples: - >>> solar_area_single_reference_year( ... start_year=2022, ... end_year=2024, @@ -976,7 +969,6 @@ def solar_area_single_reference_year( Returns: pd.DataFrame with columns datetime and value """ - return get_zone_single_reference_year( start_year=start_year, end_year=end_year, @@ -996,13 +988,11 @@ def solar_area_multiple_reference_years( directory: str | Path, year_type: Literal["fy", "calendar"] = "fy", ) -> pd.DataFrame: - """ - Pass-through function to keep backwards capability with previos API + """Pass-through function to keep backwards capability with previos API. Reads solar area trace data from an output directory created by isp_trace_parser.solar_trace_parser. Examples: - >>> solar_area_multiple_reference_years( ... reference_years={2022: 2011, 2024: 2012}, ... area='Q1', @@ -1037,8 +1027,8 @@ def solar_area_multiple_reference_years( FY2015/2016). If 'calendar', then filtering is by calendar year. Returns: pd.DataFrame with columns datetime and value - """ + """ return get_zone_multiple_reference_years( reference_year_mapping=reference_years, zone=area, @@ -1055,13 +1045,11 @@ def wind_project_multiple_reference_years( directory: str | Path, year_type: Literal["fy", "calendar"] = "fy", ) -> pd.DataFrame: - """ - Pass-through function to keep backwards capability with previos API + """Pass-through function to keep backwards capability with previos API. Reads wind project trace data from an output directory created by isp_trace_parser.wind_trace_parser. Examples: - >>> wind_project_multiple_reference_years( ... reference_years={2022: 2011, 2024: 2012}, ... project='Bango 973 Wind Farm', @@ -1094,8 +1082,8 @@ def wind_project_multiple_reference_years( FY2015/2016). If 'calendar', then filtering is by calendar year. Returns: pd.DataFrame with columns datetime and value - """ + """ return get_project_multiple_reference_years( reference_year_mapping=reference_years, project=project, @@ -1114,12 +1102,11 @@ def wind_area_single_reference_year( directory: str | Path, year_type: Literal["fy", "calendar"] = "fy", ) -> pd.DataFrame: - """ - Pass-through function to keep backwards capability with previos API + """Pass-through function to keep backwards capability with previous API. + Reads wind area trace data from an output directory created by isp_trace_parser.wind_trace_parser. Examples: - >>> wind_area_single_reference_year( ... start_year=2022, ... end_year=2024, @@ -1157,8 +1144,8 @@ def wind_area_single_reference_year( FY2015/2016). If 'calendar', then filtering is by calendar year. Returns: pd.DataFrame with columns datetime and value - """ + """ return get_zone_single_reference_year( start_year=start_year, end_year=end_year, @@ -1179,12 +1166,11 @@ def demand_multiple_reference_years( directory: str | Path, year_type: Literal["fy", "calendar"] = "fy", ) -> pd.DataFrame: - """Pass-through function to keep backwards capability with previos API + """Pass-through function to keep backwards capability with previos API. Reads wind area trace data from an output directory created by isp_trace_parser.demand_trace_parser. Examples: - >>> demand_multiple_reference_years( ... reference_years={2024: 2011}, ... subregion='CNSW', @@ -1223,8 +1209,8 @@ def demand_multiple_reference_years( FY2015/2016). If 'calendar', then filtering is by calendar year. Returns: pd.DataFrame with columns datetime and value - """ + """ return get_demand_multiple_reference_years( reference_year_mapping=reference_years, scenario=scenario, @@ -1244,11 +1230,9 @@ def wind_area_multiple_reference_years( directory: str | Path, year_type: Literal["fy", "calendar"] = "fy", ) -> pd.DataFrame: - """ - Reads wind area trace data from an output directory created by isp_trace_parser.restructure_solar_directory. + """Read wind area trace data from an output directory created by isp_trace_parser.restructure_solar_directory. Examples: - >>> wind_area_multiple_reference_years( ... reference_years={2022: 2011, 2024: 2012}, ... area='Q1', @@ -1283,8 +1267,8 @@ def wind_area_multiple_reference_years( FY2015/2016). If 'calendar', then filtering is by calendar year. Returns: pd.DataFrame with columns datetime and value - """ + """ return get_zone_multiple_reference_years( reference_year_mapping=reference_years, zone=area, @@ -1306,13 +1290,11 @@ def demand_single_reference_year( directory: str | Path, year_type: Literal["fy", "calendar"] = "fy", ) -> pd.DataFrame: - """ - Pass-through function to keep backwards capability with previos API + """Pass-through function to keep backwards capability with previos API. Reads demand trace data from an output directory created by isp_trace_parser.demand_trace_parser. Examples: - >>> demand_single_reference_year( ... start_year=2024, ... end_year=2024, @@ -1354,8 +1336,8 @@ def demand_single_reference_year( FY2015/2016). If 'calendar', then filtering is by calendar year. Returns: pd.DataFrame with columns datetime and value - """ + """ return get_demand_single_reference_year( start_year=start_year, end_year=end_year, diff --git a/src/isp_trace_parser/mappings/__init__.py b/src/isp_trace_parser/mappings/__init__.py index 4813671..1afc38b 100644 --- a/src/isp_trace_parser/mappings/__init__.py +++ b/src/isp_trace_parser/mappings/__init__.py @@ -19,6 +19,7 @@ def load(name: str, version: str = "2024") -> dict: Returns: Parsed YAML contents. + """ resource = files(__package__).joinpath(version, f"{name}.yaml") with resource.open("r") as f: diff --git a/src/isp_trace_parser/optimise_parquet.py b/src/isp_trace_parser/optimise_parquet.py index a549377..609de60 100644 --- a/src/isp_trace_parser/optimise_parquet.py +++ b/src/isp_trace_parser/optimise_parquet.py @@ -17,6 +17,7 @@ def _delete_source_files(input_directory: str | Path) -> None: Args: input_directory: Directory containing parquet files to delete + """ input_path = Path(input_directory) files = list(input_path.rglob("*.parquet")) @@ -58,8 +59,8 @@ def partition_traces_by_columns( ... "optimized_demand/", ... partition_cols=["scenario", "reference_year"] ... ) # doctest: +SKIP - """ + """ if sort_by is None: # Avoid use of mutable data structure for argument defaults # (see Ruff rule B006). diff --git a/src/isp_trace_parser/remote/download.py b/src/isp_trace_parser/remote/download.py index d16bfbd..ae62a3b 100644 --- a/src/isp_trace_parser/remote/download.py +++ b/src/isp_trace_parser/remote/download.py @@ -56,6 +56,7 @@ def _download_from_manifest( If any download fails OSError If there are filesystem errors (permissions, disk space, etc.) + """ # Construct manifest path manifest_path = files("isp_trace_parser.remote.manifests") / f"{manifest_name}.txt" @@ -121,6 +122,7 @@ def _download_file( If the download fails OSError If there are filesystem errors + """ # Parse URL to extract path parsed_url = urlparse(url) @@ -217,8 +219,8 @@ def fetch_trace_data( >>> fetch_trace_data("full", "isp_2024", "data/archive", "archive") # doctest: +SKIP # Downloads original zip files to: data/archive/... - """ + """ # Validate inputs if dataset_type not in ["full", "example"]: msg = f"dataset_type must be 'full' or 'example', got: {dataset_type}" diff --git a/src/isp_trace_parser/resource_trace_metadata.py b/src/isp_trace_parser/resource_trace_metadata.py index 8cedc7a..7f92a09 100644 --- a/src/isp_trace_parser/resource_trace_metadata.py +++ b/src/isp_trace_parser/resource_trace_metadata.py @@ -33,7 +33,6 @@ def build( The mapping key is the trace stem (the filename with `_RefYear.csv` stripped) so `_RefYear.csv` decomposes back to (stem, year). """ - resource_mapping = mappings.load("resources", version=version) file_metadata: dict[Path, dict[str, str]] = {} diff --git a/src/isp_trace_parser/solar_traces.py b/src/isp_trace_parser/solar_traces.py index f858fdb..f9bf9c6 100644 --- a/src/isp_trace_parser/solar_traces.py +++ b/src/isp_trace_parser/solar_traces.py @@ -35,7 +35,6 @@ class SolarMetadataFilter(BaseModel): included then only traces with metadata matching the values in the corresponding list will be parsed. Examples: - Filter for only projects or zones that are in a list of names. >>> metadata_filters = SolarMetadataFilter( @@ -54,6 +53,7 @@ class SolarMetadataFilter(BaseModel): file_type: list of 'project' and/or 'zone' (zone typically refers to REZs) resource_type: list of resource types of traces, only including 'SAT', 'FFP', or 'CST'. reference_year: list of ints specifying reference_years + """ name: list[str] | None = None @@ -69,7 +69,7 @@ def parse_solar_traces( use_concurrency: bool = True, filters: SolarMetadataFilter | None = None, ) -> None: - """Takes a directory with AEMO solar trace data and reformats the data, saving it to a new directory. + """Take a directory with AEMO solar trace data and reformats the data, saving it to a new directory. AEMO solar trace data comes in CSVs with columns specifying the year, day, and month, and data columns (labeled 01, 02, ... 48) storing the solar generation values for each half hour of the day. The file name of the CSV @@ -97,7 +97,6 @@ def parse_solar_traces( metadata value present in the corresponding filter list will be parsed, see examples below. Examples: - Parse whole directory of trace data. >>> parse_solar_traces( @@ -135,6 +134,7 @@ def parse_solar_traces( attribute is not set, no filtering on that attribute occurs. See example. Returns: None + """ input_directory = input_validation.input_directory(input_directory) parsed_directory = input_validation.parsed_directory(parsed_directory) @@ -196,8 +196,7 @@ def restructure_solar_files( output_directory: str | Path, filters: SolarMetadataFilter = None, ) -> None: - """ - Restructures solar trace files and saves them in a new format. + """Restructures solar trace files and saves them in a new format. This function processes solar trace files, restructures them based on the provided metadata, and saves them in a new format. It handles both project and zone solar trace files. @@ -226,8 +225,8 @@ def restructure_solar_files( ... ) # doctest: +SKIP # This will process 'file1.csv' and save it with the new name 'NewProject1' in the specified output directory - """ + """ metadata_for_trace_files = get_metadata_that_matches_trace_names( input_trace_names, all_input_file_metadata ) @@ -256,14 +255,14 @@ def restructure_solar_files( def write_output_solar_filename(metadata: dict[str, str]) -> str: - """ - Generates the output filename for a solar trace file. + """Generate the output filename for a solar trace file. Args: metadata: Dictionary containing metadata for the solar trace file. Returns: A string representing the filename. + """ m = metadata name = m["name"].replace(" ", "_") @@ -273,14 +272,14 @@ def write_output_solar_filename(metadata: dict[str, str]) -> str: def get_unique_resource_types_in_metadata( metadata_for_trace_files: dict[Path, dict[str, str]], ) -> list[str]: - """ - Gets unique resource types from the metadata of trace files. + """Get unique resource types from the metadata of trace files. Args: metadata_for_trace_files: Dictionary containing metadata for trace files. Returns: A list of unique resource types. + """ return list( {metadata["resource_type"] for metadata in metadata_for_trace_files.values()} @@ -290,8 +289,7 @@ def get_unique_resource_types_in_metadata( def get_metadata_that_matches_resource_type( resource_type: str, metadata_for_trace_files: dict[Path, dict[str, str]] ) -> dict[Path, dict[str, str]]: - """ - Filters metadata to only include files matching a specific resource type. + """Filter metadata to only include files matching a specific resource type. Args: resource_type: The resource type to filter by. @@ -299,6 +297,7 @@ def get_metadata_that_matches_resource_type( Returns: A dictionary of metadata for files matching the specified resource type. + """ return { f: metadata diff --git a/src/isp_trace_parser/trace_formatter.py b/src/isp_trace_parser/trace_formatter.py index c9ec713..7948fd1 100644 --- a/src/isp_trace_parser/trace_formatter.py +++ b/src/isp_trace_parser/trace_formatter.py @@ -13,8 +13,7 @@ @validate_call(config=config.ConfigDict(arbitrary_types_allowed=True)) def trace_formatter(trace_data: pl.DataFrame) -> pl.DataFrame: - """ - Takes trace data in the AEMO format and converts it to a format with 'datetime' and 'value' columns. + """Take trace data in the AEMO format and converts it to a format with 'datetime' and 'value' columns. AEMO provides ISP trace data with separate columns for 'Year', 'Month', and 'Day', and individual data columns labeled '01', '02', ..., '48', representing half-hour intervals. This function converts that data format into @@ -22,7 +21,6 @@ def trace_formatter(trace_data: pl.DataFrame) -> pl.DataFrame: the corresponding values. Example: - Input format (example): >>> aemo_format_data = pl.DataFrame({ @@ -59,8 +57,8 @@ def trace_formatter(trace_data: pl.DataFrame) -> pl.DataFrame: A `polars.DataFrame` with: - 'datetime': A column specifying the end time of each half-hour period. - 'value': A column containing the data for each half-hour period. - """ + """ # Need both padded 1-9 and not padded because AEMO data files can have both. value_vars = [f"{i:02d}" for i in range(1, 49)] + [str(i) for i in range(1, 10)] value_vars = [v for v in value_vars if v in trace_data.columns] diff --git a/src/isp_trace_parser/trace_restructure_helper_functions.py b/src/isp_trace_parser/trace_restructure_helper_functions.py index 0a74dbb..3c81538 100644 --- a/src/isp_trace_parser/trace_restructure_helper_functions.py +++ b/src/isp_trace_parser/trace_restructure_helper_functions.py @@ -41,13 +41,10 @@ def calculate_average_trace(traces: list[pl.DataFrame]) -> pl.DataFrame: def _frame_with_metadata(trace: pl.DataFrame, file_metadata: dict) -> pl.DataFrame: - """ - Adds metadata fields as columns to a resource trace DataFrame. + """Add metadata fields as columns to a resource trace DataFrame. Name column dynamically named based on "file_type" (ie. project or zone) - """ - return trace.with_columns( pl.lit(file_metadata["name"]).alias(file_metadata["file_type"]), pl.lit(file_metadata["reference_year"]).alias("reference_year"), diff --git a/src/isp_trace_parser/wind_traces.py b/src/isp_trace_parser/wind_traces.py index 2ff7467..241b111 100644 --- a/src/isp_trace_parser/wind_traces.py +++ b/src/isp_trace_parser/wind_traces.py @@ -35,7 +35,6 @@ class WindMetadataFilter(BaseModel): included then only traces with metadata matching the values in the corresponding list will be parsed. Examples: - Filter for only projects or zones that are in a list of names. >>> metadata_filters = WindMetadataFilter( @@ -54,6 +53,7 @@ class WindMetadataFilter(BaseModel): file_type: list of 'project' and/or 'zone' (zone typically refers to REZs) resource_type: list of resource types, only including 'WH', 'WM', 'WL', 'WX', or 'wind'. reference_year: list of ints specifying reference_years + """ name: list[str] | None = None @@ -69,7 +69,7 @@ def parse_wind_traces( use_concurrency: bool = True, filters: WindMetadataFilter | None = None, ) -> None: - """Takes a directory with AEMO wind trace data and reformats the data, saving it to a new directory. + """Take a directory with AEMO wind trace data and reformats the data, saving it to a new directory. AEMO wind trace data comes in CSVs with columns specifying the year, day, and month, and data columns (labeled 01, 02, ... 48) storing the wind generation values for each half hour of the day. The file name of the CSV @@ -98,7 +98,6 @@ def parse_wind_traces( metadata value present in the corresponding filter list will be parsed, see examples below. Examples: - Parse whole directory of trace data. >>> parse_wind_traces( @@ -136,6 +135,7 @@ def parse_wind_traces( attribute is not set, no filtering on that attribute occurs. See example. Returns: None + """ input_directory = input_validation.input_directory(input_directory) parsed_directory = input_validation.parsed_directory(parsed_directory) @@ -222,11 +222,9 @@ def restructure_wind_zone_files( output_directory: str | Path, filters: dict[str, list[str]] | None = None, ) -> None: - """ - Restructures wind zone trace files and saves them in a new format. + """Restructures wind zone trace files and saves them in a new format. Examples: - >>> all_metadata = { ... 'file1.csv': {'name': 'Zone1', 'year': '2020', 'resource_type': 'WH'}, ... 'file2.csv': {'name': 'Zone1', 'year': '2021', 'resource_type': 'WM'}, @@ -252,6 +250,7 @@ def restructure_wind_zone_files( Returns: None: Files are saved to disk, but the function doesn't return any value. + """ metadata_for_trace_files = get_metadata_that_matches_trace_names( input_trace_names, all_input_file_metadata @@ -287,9 +286,7 @@ def restructure_wind_project_files( output_directory: str | Path, filters: dict[str, list[str]] | None = None, ) -> None: - """ - Restructures wind project trace files and saves them in a new format. - """ + """Restructures wind project trace files and saves them in a new format.""" metadata_for_trace_files = get_metadata_that_matches_trace_names( input_trace_names, all_input_file_metadata ) @@ -313,8 +310,7 @@ def restructure_wind_project_files( def write_output_wind_project_filename(metadata: dict) -> str: - """ - Generates the output filename for a wind project trace file. + """Generate the output filename for a wind project trace file. Returns a string representing the filename. """ @@ -324,8 +320,7 @@ def write_output_wind_project_filename(metadata: dict) -> str: def write_output_wind_zone_filename(metadata: dict) -> str: - """ - Generates the output filename for a wind zone trace file. + """Generate the output filename for a wind zone trace file. Returns a string representing the filename. """