diff --git a/.flake8 b/.flake8 deleted file mode 100644 index 52471ad28..000000000 --- a/.flake8 +++ /dev/null @@ -1,7 +0,0 @@ -[flake8] -; https://github.com/PyCQA/flake8 -max-line-length=100 -select=C,E,F,W,B,B950 -ignore=E203,E501,W503,F401 -exclude=.git,__pycache__,docs/source/conf.py,build,dist,paper -docstring-convention=google diff --git a/.git-blame-ignore-revs b/.git-blame-ignore-revs new file mode 100644 index 000000000..306a22a53 --- /dev/null +++ b/.git-blame-ignore-revs @@ -0,0 +1,3 @@ +# Migrate from Black to ruff +bc2e00b5eaee49cca7829dd615dedbf9d202fa11 +df59900337de9c74cdbc7aba8511e3840afb0737 diff --git a/.pre-commit-config.yaml b/.pre-commit-config.yaml index d9423c9bc..9ecd18841 100644 --- a/.pre-commit-config.yaml +++ b/.pre-commit-config.yaml @@ -1,14 +1,6 @@ repos: - - repo: https://github.com/psf/black - rev: 26.1.0 - hooks: - - id: black - name: black - stages: [pre-commit] - language_version: python3 - - repo: https://github.com/pycqa/isort - rev: 7.0.0 + rev: 9.0.2 hooks: - id: isort additional_dependencies: [toml] @@ -22,11 +14,14 @@ repos: additional_dependencies: [toml] types: [pyi] - - repo: https://github.com/asottile/pyupgrade - rev: v3.21.2 + - repo: https://github.com/astral-sh/ruff-pre-commit + rev: v0.16.10 hooks: - - id: pyupgrade - args: [--py38-plus] + - id: ruff-check + types_or: [python, pyi, jupyter] + args: [--fix] + - id: ruff-format + types_or: [python, pyi, jupyter] - repo: https://github.com/pre-commit/pre-commit-hooks rev: v6.0.0 @@ -40,8 +35,3 @@ repos: - id: mixed-line-ending - id: pretty-format-json args: [--autofix] - - - repo: https://github.com/pycqa/flake8 - rev: '7.3.0' - hooks: - - id: flake8 diff --git a/CHANGELOG.md b/CHANGELOG.md index f08a9f7b2..b3493d1db 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,6 +1,10 @@ # Changelog All notable changes to this project will be documented in this file. If you make a notable change to the project, please add a line describing the change to the "unreleased" section. The maintainers will make an effort to keep the [Github Releases](https://github.com/NREL/OpenOA/releases) page up to date with this changelog. The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/). +## Unreleased + +- Replace `black` and `flake8` with `ruff` for faster and more robust linting/automated formatting. + ## v3.2 - 2026-01-29 - Features and updates: diff --git a/README.md b/README.md index ed3742187..c9fee3e21 100644 --- a/README.md +++ b/README.md @@ -13,7 +13,7 @@ [![PyPI downloads](https://img.shields.io/pypi/dm/openoa)](https://pypi.org/project/WOMBAT/) [![pre-commit](https://img.shields.io/badge/pre--commit-enabled-brightgreen?logo=pre-commit&logoColor=white)](https://github.com/pre-commit/pre-commit) -[![Code style: black](https://img.shields.io/badge/code%20style-black-000000.svg)](https://github.com/psf/black) +[![Code style: Ruff](https://img.shields.io/endpoint?url=https://raw.githubusercontent.com/astral-sh/ruff/main/assets/badge/v2.json)](https://github.com/astral-sh/ruff) [![Imports: isort](https://img.shields.io/badge/%20imports-isort-%231674b1?style=flat&labelColor=ef8336)](https://pycqa.github.io/isort/) [![All Contributors](https://img.shields.io/badge/all_contributors-15-orange.svg?style=flat-square)](#contributors-) diff --git a/contributing.md b/contributing.md index a8b91c990..d6b4029c0 100644 --- a/contributing.md +++ b/contributing.md @@ -143,8 +143,8 @@ will need to accept the Contributor License Agreement(CLA). ## Coding Style This code uses a ``pre-commit`` workflow where code styling and linting is taken care of when a user -commits their code. Specifically, this code utilizes ``black`` for automatic formatting (line length, quotation usage, hanging -lines, etc.), ``isort`` for automatic import sorting, and ``flake8`` for linting. +commits their code. Specifically, this code utilizes ``ruff`` for automatic formatting (line length, +quotation usage, hanging lines, etc.) and linting and ``isort`` for automatic import sorting. To activate the ``pre-commit`` workflow, the user must install the develop version as outlined in the [Readme](https://github.com/NREL/OpenOA/tree/develop#Development), and run the following line: diff --git a/examples/00_intro_to_plant_data.ipynb b/examples/00_intro_to_plant_data.ipynb index 11758d8f7..378ce7066 100644 --- a/examples/00_intro_to_plant_data.ipynb +++ b/examples/00_intro_to_plant_data.ipynb @@ -809,7 +809,7 @@ " df=scada_df_tz,\n", " time_col=\"Date_time\",\n", " local_tz=\"Europe/Paris\",\n", - " tz_aware=True # Indicate that we can use encoded data to convert between timezones\n", + " tz_aware=True, # Indicate that we can use encoded data to convert between timezones\n", ")\n", "scada_df_tz.head()" ] @@ -1043,7 +1043,7 @@ " df=scada_df_no_tz,\n", " time_col=\"Date_time\",\n", " local_tz=\"Europe/Paris\",\n", - " tz_aware=False # Indicates that we're going to need to make inferences about encoding the timezones\n", + " tz_aware=False, # Indicates that we're going to need to make inferences about encoding the timezones\n", ")\n", "scada_df_no_tz.head()" ] @@ -1249,8 +1249,19 @@ ], "source": [ "no_tz = qa.describe(scada_df_no_tz)\n", - "no_tz = no_tz.loc[~no_tz.index.isin([\"Date_time\"])] # Ignore the Date_time column that is not shared between the dataframes\n", - "col_order = [\"count\", \"mean\", \"std\", \"min\", \"25%\", \"50%\", \"75%\", \"max\"] # Ensure description columns are in the same order\n", + "no_tz = no_tz.loc[\n", + " ~no_tz.index.isin([\"Date_time\"])\n", + "] # Ignore the Date_time column that is not shared between the dataframes\n", + "col_order = [\n", + " \"count\",\n", + " \"mean\",\n", + " \"std\",\n", + " \"min\",\n", + " \"25%\",\n", + " \"50%\",\n", + " \"75%\",\n", + " \"max\",\n", + "] # Ensure description columns are in the same order\n", "qa.describe(scada_df_tz)[col_order] == no_tz[col_order]" ] }, @@ -1621,9 +1632,7 @@ ], "source": [ "dup_orig_no_tz, dup_local_no_tz, dup_utc_no_tz = qa.duplicate_time_identification(\n", - " df=scada_df_no_tz,\n", - " time_col=\"Date_time\",\n", - " id_col=\"Wind_turbine_name\"\n", + " df=scada_df_no_tz, time_col=\"Date_time\", id_col=\"Wind_turbine_name\"\n", ")\n", "dup_orig_no_tz.size, dup_local_no_tz.size, dup_utc_no_tz.size" ] @@ -1729,9 +1738,7 @@ ], "source": [ "gap_orig_no_tz, gap_local_no_tz, gap_utc_no_tz = qa.gap_time_identification(\n", - " df=scada_df_no_tz,\n", - " time_col=\"Date_time\",\n", - " freq=\"10min\"\n", + " df=scada_df_no_tz, time_col=\"Date_time\", freq=\"10min\"\n", ")\n", "gap_orig_no_tz.size, gap_local_no_tz.size, gap_utc_no_tz.size" ] @@ -1840,7 +1847,7 @@ " time_col=\"Date_time\",\n", " power_col=\"P_avg\",\n", " freq=\"10min\",\n", - " hour_window=3 # default value\n", + " hour_window=3, # default value\n", ")" ] }, @@ -1864,9 +1871,7 @@ "outputs": [], "source": [ "dup_orig_tz, dup_local_tz, dup_utc_tz = qa.duplicate_time_identification(\n", - " df=scada_df_tz,\n", - " time_col=\"Date_time\",\n", - " id_col=\"Wind_turbine_name\"\n", + " df=scada_df_tz, time_col=\"Date_time\", id_col=\"Wind_turbine_name\"\n", ")" ] }, @@ -1980,9 +1985,7 @@ ], "source": [ "gap_orig_tz, gap_local_tz, gap_utc_tz = qa.gap_time_identification(\n", - " df=scada_df_tz,\n", - " time_col=\"Date_time\",\n", - " freq=\"10min\"\n", + " df=scada_df_tz, time_col=\"Date_time\", freq=\"10min\"\n", ")\n", "gap_orig_tz.size, gap_local_tz.size, gap_utc_tz.size" ] @@ -2054,7 +2057,7 @@ " time_col=\"Date_time\",\n", " power_col=\"P_avg\",\n", " freq=\"10min\",\n", - " hour_window=3 # default value\n", + " hour_window=3, # default value\n", ")" ] }, @@ -2265,6 +2268,7 @@ ], "source": [ "from openoa.plant import PlantMetaData\n", + "\n", "print(PlantMetaData.__doc__)" ] }, @@ -2349,7 +2353,7 @@ "print(f\"The new analysis types now has all and MonteCarloAEP: {engie.analysis_type}\")\n", "try:\n", " engie.validate()\n", - "except ValueError as e: # Catch the error message so that the whole notebook can run\n", + "except ValueError as e: # Catch the error message so that the whole notebook can run\n", " print(e)" ] }, @@ -2398,7 +2402,7 @@ " asset=f\"{data_path}/asset.csv\",\n", " reanalysis={\n", " \"era5\": f\"{data_path}/reanalysis_era5.csv\",\n", - " \"merra2\": f\"{data_path}/reanalysis_merra2.csv\"\n", + " \"merra2\": f\"{data_path}/reanalysis_merra2.csv\",\n", " },\n", ")" ] diff --git a/examples/01_utils_examples.ipynb b/examples/01_utils_examples.ipynb index f3c7e26ae..5a97d4267 100644 --- a/examples/01_utils_examples.ipynb +++ b/examples/01_utils_examples.ipynb @@ -367,6 +367,7 @@ "\n", "from bokeh.plotting import show\n", "from bokeh.io import output_notebook\n", + "\n", "output_notebook()\n", "\n", "from openoa.utils import filters, power_curve, plot\n", @@ -697,7 +698,7 @@ " xlim=(-1, 20), # optional input for refining plots\n", " ylim=(-100, 2100), # optional input for refining plots\n", " legend=True, # optional flag for adding a legend\n", - " scatter_kwargs=dict(alpha=0.8, s=10) # optional input for refining plots\n", + " scatter_kwargs=dict(alpha=0.8, s=10), # optional input for refining plots\n", ")" ] }, @@ -757,7 +758,7 @@ } ], "source": [ - "out_of_window = filters.window_range_flag(windspeed, 5., 40, power_kw, 20., 2100.)\n", + "out_of_window = filters.window_range_flag(windspeed, 5.0, 40, power_kw, 20.0, 2100.0)\n", "plot.plot_power_curve(\n", " windspeed,\n", " power_kw,\n", @@ -766,7 +767,7 @@ " xlim=(-1, 20), # optional input for refining plots\n", " ylim=(-100, 2100), # optional input for refining plots\n", " legend=True, # optional flag for adding a legend\n", - " scatter_kwargs=dict(alpha=0.4, s=10) # optional input for refining plots\n", + " scatter_kwargs=dict(alpha=0.4, s=10), # optional input for refining plots\n", ")" ] }, @@ -816,7 +817,9 @@ ], "source": [ "max_bin = 0.90 * power_kw_filt1.max()\n", - "bin_outliers = filters.bin_filter(power_kw_filt1, windspeed_filt1, 100, 1.5, \"median\", 20., max_bin, \"scalar\", \"all\")\n", + "bin_outliers = filters.bin_filter(\n", + " power_kw_filt1, windspeed_filt1, 100, 1.5, \"median\", 20.0, max_bin, \"scalar\", \"all\"\n", + ")\n", "plot.plot_power_curve(\n", " windspeed_filt1,\n", " power_kw_filt1,\n", @@ -825,7 +828,7 @@ " xlim=(-1, 20), # optional input for refining plots\n", " ylim=(-100, 2100), # optional input for refining plots\n", " legend=True, # optional flag for adding a legend\n", - " scatter_kwargs=dict(alpha=0.5, s=10) # optional input for refining plots\n", + " scatter_kwargs=dict(alpha=0.5, s=10), # optional input for refining plots\n", ")" ] }, @@ -881,7 +884,7 @@ " xlim=(-1, 20), # optional input for refining plots\n", " ylim=(-100, 2100), # optional input for refining plots\n", " legend=True, # optional flag for adding a legend\n", - " scatter_kwargs=dict(alpha=0.4, s=10) # optional input for refining plots\n", + " scatter_kwargs=dict(alpha=0.4, s=10), # optional input for refining plots\n", ")" ] }, @@ -972,9 +975,9 @@ ")\n", "\n", "x = np.linspace(0, 20, 100)\n", - "ax.plot(x, iec_curve(x), color=\"red\", label = \"IEC\", linewidth = 3)\n", - "ax.plot(x, spline_curve(x), color=\"C1\", label = \"Spline\", linewidth = 3)\n", - "ax.plot(x, l5p_curve(x), color=\"C2\", label = \"L5P\", linewidth = 3)\n", + "ax.plot(x, iec_curve(x), color=\"red\", label=\"IEC\", linewidth=3)\n", + "ax.plot(x, spline_curve(x), color=\"C1\", label=\"Spline\", linewidth=3)\n", + "ax.plot(x, l5p_curve(x), color=\"C2\", label=\"L5P\", linewidth=3)\n", "\n", "ax.legend()\n", "\n", diff --git a/examples/02a_plant_aep_analysis.ipynb b/examples/02a_plant_aep_analysis.ipynb index b388fae6b..15a7ba087 100644 --- a/examples/02a_plant_aep_analysis.ipynb +++ b/examples/02a_plant_aep_analysis.ipynb @@ -71,7 +71,7 @@ "outputs": [], "source": [ "# Load plant object\n", - "project = project_ENGIE.prepare('./data/la_haute_borne', use_cleansed=False)" + "project = project_ENGIE.prepare(\"./data/la_haute_borne\", use_cleansed=False)" ] }, { @@ -213,7 +213,7 @@ "metadata": {}, "outputs": [], "source": [ - "pa = MonteCarloAEP(project, reanalysis_products = ['era5', 'merra2'])" + "pa = MonteCarloAEP(project, reanalysis_products=[\"era5\", \"merra2\"])" ] }, { @@ -520,7 +520,9 @@ } ], "source": [ - "pa.plot_reanalysis_gross_energy_data(outlier_threshold=3, xlim=(4, 9), ylim=(0, 2), plot_kwargs=dict(s=60))" + "pa.plot_reanalysis_gross_energy_data(\n", + " outlier_threshold=3, xlim=(4, 9), ylim=(0, 2), plot_kwargs=dict(s=60)\n", + ")" ] }, { @@ -550,9 +552,7 @@ ], "source": [ "pa.plot_aggregate_plant_data_timeseries(\n", - " xlim=(datetime(2013, 12, 1), datetime(2015, 12, 31)),\n", - " ylim_energy=(0, 2),\n", - " ylim_loss=(-0.1, 5.5)\n", + " xlim=(datetime(2013, 12, 1), datetime(2015, 12, 31)), ylim_energy=(0, 2), ylim_loss=(-0.1, 5.5)\n", ")" ] }, @@ -576,8 +576,8 @@ "outputs": [], "source": [ "# For illustrative purposes, let's suppose a few months aren't representative of long-term losses\n", - "pa.aggregate.loc['2014-11-01',['availability_typical','curtailment_typical']] = False\n", - "pa.aggregate.loc['2015-07-01',['availability_typical','curtailment_typical']] = False" + "pa.aggregate.loc[\"2014-11-01\", [\"availability_typical\", \"curtailment_typical\"]] = False\n", + "pa.aggregate.loc[\"2015-07-01\", [\"availability_typical\", \"curtailment_typical\"]] = False" ] }, { @@ -654,7 +654,7 @@ ], "source": [ "# Run Monte Carlo based OA\n", - "pa.run(num_sim=2000, reanalysis_products=['era5', 'merra2'])" + "pa.run(num_sim=2000, reanalysis_products=[\"era5\", \"merra2\"])" ] }, { @@ -729,16 +729,18 @@ "outputs": [], "source": [ "# Produce histograms of the various MC-parameters\n", - "mc_reg = pd.DataFrame(data={\n", - " 'slope': pa._mc_slope.ravel(),\n", - " 'intercept': pa._mc_intercept, \n", - " 'num_points': pa._mc_num_points, \n", - " 'metered_energy_fraction': pa.mc_inputs.metered_energy_fraction, \n", - " 'loss_fraction': pa.mc_inputs.loss_fraction,\n", - " 'num_years_windiness': pa.mc_inputs.num_years_windiness, \n", - " 'loss_threshold': pa.mc_inputs.loss_threshold,\n", - " 'reanalysis_product': pa.mc_inputs.reanalysis_product\n", - "})" + "mc_reg = pd.DataFrame(\n", + " data={\n", + " \"slope\": pa._mc_slope.ravel(),\n", + " \"intercept\": pa._mc_intercept,\n", + " \"num_points\": pa._mc_num_points,\n", + " \"metered_energy_fraction\": pa.mc_inputs.metered_energy_fraction,\n", + " \"loss_fraction\": pa.mc_inputs.loss_fraction,\n", + " \"num_years_windiness\": pa.mc_inputs.num_years_windiness,\n", + " \"loss_threshold\": pa.mc_inputs.loss_threshold,\n", + " \"reanalysis_product\": pa.mc_inputs.reanalysis_product,\n", + " }\n", + ")" ] }, { @@ -800,17 +802,18 @@ "# Produce scatter plots of slope and intercept values. Here we focus on the ERA-5 data\n", "plot.set_styling()\n", "\n", - "plt.figure(figsize=(8,6))\n", + "plt.figure(figsize=(8, 6))\n", "plt.plot(\n", - " mc_reg.intercept[mc_reg.reanalysis_product =='era5'],\n", - " mc_reg.slope[mc_reg.reanalysis_product =='era5'],\n", - " '.', label=\"Monte Carlo Regression Values\"\n", + " mc_reg.intercept[mc_reg.reanalysis_product == \"era5\"],\n", + " mc_reg.slope[mc_reg.reanalysis_product == \"era5\"],\n", + " \".\",\n", + " label=\"Monte Carlo Regression Values\",\n", ")\n", "x = np.linspace(-2, 0, 3)\n", "y = -0.2 * x + 0.135\n", "plt.plot(x, y, label=\"y = -0.2x + 0.135\")\n", - "plt.xlabel('Intercept (GWh)')\n", - "plt.ylabel('Slope (GWh / (m/s))')\n", + "plt.xlabel(\"Intercept (GWh)\")\n", + "plt.ylabel(\"Slope (GWh / (m/s))\")\n", "plt.legend()\n", "plt.xlim((-1.8, -0.4))\n", "plt.ylim(0.2, 0.5)\n", @@ -841,7 +844,7 @@ } ], "source": [ - "pa.plot_aep_boxplot(x=mc_reg['reanalysis_product'], xlabel=\"Reanalysis Product\", ylim=(6, 18))" + "pa.plot_aep_boxplot(x=mc_reg[\"reanalysis_product\"], xlabel=\"Reanalysis Product\", ylim=(6, 18))" ] }, { @@ -874,11 +877,11 @@ "# NOTE: This is the same method, but calling the same method through the plot module directly\n", "plot.plot_boxplot(\n", " y=pa.results.aep_GWh,\n", - " x=mc_reg['num_years_windiness'],\n", + " x=mc_reg[\"num_years_windiness\"],\n", " xlabel=\"Number of Years in the Windiness Correction\",\n", " ylabel=\"AEP (GWh/yr)\",\n", " ylim=(0, 20),\n", - " plot_kwargs_box={\"flierprops\":dict(marker=\"x\", markeredgecolor=\"tab:blue\")}\n", + " plot_kwargs_box={\"flierprops\": dict(marker=\"x\", markeredgecolor=\"tab:blue\")},\n", ")" ] }, @@ -909,13 +912,13 @@ ], "source": [ "fig, ax, boxes = pa.plot_aep_boxplot(\n", - " x=mc_reg['num_years_windiness'],\n", + " x=mc_reg[\"num_years_windiness\"],\n", " xlabel=\"Number of Years in the Windiness Correction\",\n", " ylim=(0, 20),\n", " figure_kwargs=dict(figsize=(12, 6)),\n", " plot_kwargs_box={\n", " \"flierprops\": dict(marker=\"x\", markeredgecolor=\"tab:blue\"),\n", - " \"medianprops\": dict(linewidth=1.5)\n", + " \"medianprops\": dict(linewidth=1.5),\n", " },\n", " return_fig=True,\n", " with_points=True,\n", diff --git a/examples/02b_plant_aep_analysis_cubico.ipynb b/examples/02b_plant_aep_analysis_cubico.ipynb index 7ecba5530..12971d42f 100644 --- a/examples/02b_plant_aep_analysis_cubico.ipynb +++ b/examples/02b_plant_aep_analysis_cubico.ipynb @@ -83,7 +83,7 @@ "metadata": {}, "outputs": [], "source": [ - "asset = \"kelmarsh\" # kelmarsh or penmanshiel\n", + "asset = \"kelmarsh\" # kelmarsh or penmanshiel\n", "\n", "project = project_Cubico.prepare(asset=asset)" ] @@ -125,7 +125,9 @@ ], "source": [ "plot.column_histograms(project.meter.resample(\"1h\").sum(numeric_only=True), columns=[\"MMTR_SupWh\"])\n", - "plot.column_histograms(project.curtail.resample(\"1h\").sum(numeric_only=True), columns=[\"IAVL_DnWh\", \"IAVL_ExtPwrDnWh\"])" + "plot.column_histograms(\n", + " project.curtail.resample(\"1h\").sum(numeric_only=True), columns=[\"IAVL_DnWh\", \"IAVL_ExtPwrDnWh\"]\n", + ")" ] }, { @@ -267,7 +269,7 @@ "metadata": {}, "outputs": [], "source": [ - "pa = MonteCarloAEP(project, reanalysis_products = list(project.reanalysis.keys()))" + "pa = MonteCarloAEP(project, reanalysis_products=list(project.reanalysis.keys()))" ] }, { @@ -728,8 +730,8 @@ "source": [ "pa.plot_normalized_monthly_reanalysis_windspeed(\n", " return_fig=False,\n", - " #xlim=(datetime(2000, 1, 1), datetime(2022, 12, 31)),\n", - " #ylim=(0.8, 1.2),\n", + " # xlim=(datetime(2000, 1, 1), datetime(2022, 12, 31)),\n", + " # ylim=(0.8, 1.2),\n", ")" ] }, @@ -821,15 +823,15 @@ "source": [ "# For both assets there are a few months that aren't representative of long-term losses, so these are excluded.\n", "\n", - "if asset == 'kelmarsh':\n", - " pa.aggregate.loc['2016-02-01',['availability_typical','curtailment_typical']] = False\n", - " pa.aggregate.loc['2016-03-01',['availability_typical','curtailment_typical']] = False\n", - " pa.aggregate.loc['2016-04-01',['availability_typical','curtailment_typical']] = False\n", - " \n", - "elif asset == 'penmanshiel':\n", - " pa.aggregate.loc['2018-01-01',['availability_typical','curtailment_typical']] = False\n", - " pa.aggregate.loc['2018-02-01',['availability_typical','curtailment_typical']] = False\n", - " pa.aggregate.loc['2018-03-01',['availability_typical','curtailment_typical']] = False\n" + "if asset == \"kelmarsh\":\n", + " pa.aggregate.loc[\"2016-02-01\", [\"availability_typical\", \"curtailment_typical\"]] = False\n", + " pa.aggregate.loc[\"2016-03-01\", [\"availability_typical\", \"curtailment_typical\"]] = False\n", + " pa.aggregate.loc[\"2016-04-01\", [\"availability_typical\", \"curtailment_typical\"]] = False\n", + "\n", + "elif asset == \"penmanshiel\":\n", + " pa.aggregate.loc[\"2018-01-01\", [\"availability_typical\", \"curtailment_typical\"]] = False\n", + " pa.aggregate.loc[\"2018-02-01\", [\"availability_typical\", \"curtailment_typical\"]] = False\n", + " pa.aggregate.loc[\"2018-03-01\", [\"availability_typical\", \"curtailment_typical\"]] = False" ] }, { @@ -975,16 +977,18 @@ "outputs": [], "source": [ "# Produce histograms of the various MC-parameters\n", - "mc_reg = pd.DataFrame(data={\n", - " 'slope': pa._mc_slope.ravel(),\n", - " 'intercept': pa._mc_intercept, \n", - " 'num_points': pa._mc_num_points, \n", - " 'metered_energy_fraction': pa.mc_inputs.metered_energy_fraction, \n", - " 'loss_fraction': pa.mc_inputs.loss_fraction,\n", - " 'num_years_windiness': pa.mc_inputs.num_years_windiness, \n", - " 'loss_threshold': pa.mc_inputs.loss_threshold,\n", - " 'reanalysis_product': pa.mc_inputs.reanalysis_product\n", - "})" + "mc_reg = pd.DataFrame(\n", + " data={\n", + " \"slope\": pa._mc_slope.ravel(),\n", + " \"intercept\": pa._mc_intercept,\n", + " \"num_points\": pa._mc_num_points,\n", + " \"metered_energy_fraction\": pa.mc_inputs.metered_energy_fraction,\n", + " \"loss_fraction\": pa.mc_inputs.loss_fraction,\n", + " \"num_years_windiness\": pa.mc_inputs.num_years_windiness,\n", + " \"loss_threshold\": pa.mc_inputs.loss_threshold,\n", + " \"reanalysis_product\": pa.mc_inputs.reanalysis_product,\n", + " }\n", + ")" ] }, { @@ -1043,14 +1047,15 @@ ], "source": [ "# Produce scatter plots of slope and intercept values. Here we focus on only one reanalysis data set\n", - "plt.figure(figsize=(8,6))\n", + "plt.figure(figsize=(8, 6))\n", "plt.plot(\n", - " mc_reg.intercept[mc_reg.reanalysis_product ==list(project.reanalysis.keys())[0]],\n", - " mc_reg.slope[mc_reg.reanalysis_product ==list(project.reanalysis.keys())[0]],\n", - " '.', label=\"Monte Carlo Regression Values\"\n", + " mc_reg.intercept[mc_reg.reanalysis_product == list(project.reanalysis.keys())[0]],\n", + " mc_reg.slope[mc_reg.reanalysis_product == list(project.reanalysis.keys())[0]],\n", + " \".\",\n", + " label=\"Monte Carlo Regression Values\",\n", ")\n", - "plt.xlabel('Intercept (GWh)')\n", - "plt.ylabel('Slope (GWh / (m/s))')\n", + "plt.xlabel(\"Intercept (GWh)\")\n", + "plt.ylabel(\"Slope (GWh / (m/s))\")\n", "plt.legend()\n", "plt.show()" ] @@ -1079,7 +1084,7 @@ } ], "source": [ - "pa.plot_aep_boxplot(x=mc_reg['reanalysis_product'], xlabel=\"Reanalysis Product\")" + "pa.plot_aep_boxplot(x=mc_reg[\"reanalysis_product\"], xlabel=\"Reanalysis Product\")" ] }, { @@ -1110,10 +1115,10 @@ "# NOTE: This is the same method, but calling the same method through the plot module directly\n", "plot.plot_boxplot(\n", " y=pa.results.aep_GWh,\n", - " x=mc_reg['num_years_windiness'],\n", + " x=mc_reg[\"num_years_windiness\"],\n", " xlabel=\"Number of Years in the Windiness Correction\",\n", " ylabel=\"AEP (GWh/yr)\",\n", - " plot_kwargs_box={\"flierprops\":dict(marker=\"x\", markeredgecolor=\"tab:blue\")}\n", + " plot_kwargs_box={\"flierprops\": dict(marker=\"x\", markeredgecolor=\"tab:blue\")},\n", ")" ] }, @@ -1144,12 +1149,12 @@ ], "source": [ "fig, ax, boxes = pa.plot_aep_boxplot(\n", - " x=mc_reg['num_years_windiness'],\n", + " x=mc_reg[\"num_years_windiness\"],\n", " xlabel=\"Number of Years in the Windiness Correction\",\n", " figure_kwargs=dict(figsize=(12, 6)),\n", " plot_kwargs_box={\n", " \"flierprops\": dict(marker=\"x\", markeredgecolor=\"tab:blue\"),\n", - " \"medianprops\": dict(linewidth=1.5)\n", + " \"medianprops\": dict(linewidth=1.5),\n", " },\n", " return_fig=True,\n", " with_points=True,\n", diff --git a/examples/03_turbine_ideal_energy.ipynb b/examples/03_turbine_ideal_energy.ipynb index 102730755..07a1fe444 100644 --- a/examples/03_turbine_ideal_energy.ipynb +++ b/examples/03_turbine_ideal_energy.ipynb @@ -64,7 +64,7 @@ "outputs": [], "source": [ "# Load plant object and validate for the turbine long term energy analysis type\n", - "project = project_ENGIE.prepare('./data/la_haute_borne', use_cleansed=False)\n", + "project = project_ENGIE.prepare(\"./data/la_haute_borne\", use_cleansed=False)\n", "project.analysis_type.append(\"TurbineLongTermGrossEnergy\")\n", "project.validate()" ] @@ -247,7 +247,7 @@ " UQ=False,\n", " wind_bin_threshold=2.0, # Exclude data outside 2 standard deviations of the median for each power bin\n", " max_power_filter=0.9, # Don't apply bin filter above 0.9 of turbine capacity\n", - " correction_threshold=0.9 # Set the correction threshold to 90%\n", + " correction_threshold=0.9, # Set the correction threshold to 90%\n", ")" ] }, @@ -283,12 +283,12 @@ } ], "source": [ - "# We can choose to save key plots to a file by setting enable_plotting=True and \n", - "# specifying a directory to save the images. For now we turn off this feature. \n", + "# We can choose to save key plots to a file by setting enable_plotting=True and\n", + "# specifying a directory to save the images. For now we turn off this feature.\n", "# ta.run(reanalysis_subset=['era5', 'merra2'], enable_plotting=False, plot_dir=None,\n", "# wind_bin_thresh=wind_bin_thresh, max_power_filter=max_power_filter,\n", "# correction_threshold=correction_threshold)\n", - "ta.run(reanalysis_products=['era5', 'merra2'])" + "ta.run(reanalysis_products=[\"era5\", \"merra2\"])" ] }, { @@ -468,12 +468,18 @@ "metadata": {}, "outputs": [], "source": [ - "ta=TurbineLongTermGrossEnergy(\n", + "ta = TurbineLongTermGrossEnergy(\n", " project,\n", " UQ=True, # enable uncertainty quantification\n", " num_sim=75, # number of Monte Carlo simulations to perform\n", - " wind_bin_threshold=(1, 3), # Data outside of a range of +-1 to +-3 standard deviations from the median for each power bin are discarded\n", - " max_power_filter=(0.8, 0.9), # The bin filter will be applied up to fractions of turbine capacity from 80% to 90%\n", + " wind_bin_threshold=(\n", + " 1,\n", + " 3,\n", + " ), # Data outside of a range of +-1 to +-3 standard deviations from the median for each power bin are discarded\n", + " max_power_filter=(\n", + " 0.8,\n", + " 0.9,\n", + " ), # The bin filter will be applied up to fractions of turbine capacity from 80% to 90%\n", " uncertainty_scada=0.005, # Assumed uncertainty of SCADA power data (0.5%)\n", " correction_threshold=(0.85, 0.95),\n", ")" @@ -509,9 +515,9 @@ } ], "source": [ - "# We can choose to save key plots to a file by setting enable_plotting=True and \n", - "# specifying a directory to save the images. For now we turn off this feature. \n", - "ta.run(reanalysis_products=['era5', 'merra2'])" + "# We can choose to save key plots to a file by setting enable_plotting=True and\n", + "# specifying a directory to save the images. For now we turn off this feature.\n", + "ta.run(reanalysis_products=[\"era5\", \"merra2\"])" ] }, { @@ -549,7 +555,9 @@ "print(f\"Mean long-term turbine ideal energy is {np.mean(ta.plant_gross / 1e6):,.1f} GWh/year\")\n", "\n", "# Uncertainty in long-term annual TIE for whole plant\n", - "print(f\"Uncertainty in long-term turbine ideal energy is {np.std(ta.plant_gross / 1e6):,.1f} GWh/year, or {np.std(ta.plant_gross) / np.mean(ta.plant_gross):.1%} percent\")" + "print(\n", + " f\"Uncertainty in long-term turbine ideal energy is {np.std(ta.plant_gross / 1e6):,.1f} GWh/year, or {np.std(ta.plant_gross) / np.mean(ta.plant_gross):.1%} percent\"\n", + ")" ] }, { diff --git a/examples/04_electrical_losses.ipynb b/examples/04_electrical_losses.ipynb index dc1ea260c..5a0fb751c 100644 --- a/examples/04_electrical_losses.ipynb +++ b/examples/04_electrical_losses.ipynb @@ -73,7 +73,7 @@ "outputs": [], "source": [ "# Load wind farm object, append the analysis type for this example, and revalidate the data\n", - "project = project_ENGIE.prepare('./data/la_haute_borne', use_cleansed=False)\n", + "project = project_ENGIE.prepare(\"./data/la_haute_borne\", use_cleansed=False)\n", "project.analysis_type.append(\"ElectricalLosses\")\n", "project.validate()" ] @@ -189,8 +189,7 @@ ], "source": [ "el.plot_monthly_losses(\n", - " xlim=(datetime(month=12, day=1, year=2013), datetime(month=1, day=1, year=2016)),\n", - " ylim=(0, 2.2)\n", + " xlim=(datetime(month=12, day=1, year=2013), datetime(month=1, day=1, year=2016)), ylim=(0, 2.2)\n", ")" ] }, @@ -218,11 +217,15 @@ "source": [ "# Create Electrical Loss object\n", "el = ElectricalLosses(\n", - " project, UQ = True, # enable UQ\n", - " num_sim=3000, # number of Monte Carlo simulations to perform\n", - " uncertainty_meter=0.005, # 0.5% uncertainty in meter data\n", + " project,\n", + " UQ=True, # enable UQ\n", + " num_sim=3000, # number of Monte Carlo simulations to perform\n", + " uncertainty_meter=0.005, # 0.5% uncertainty in meter data\n", " uncertainty_scada=0.005, # 0.5% uncertainty in scada data\n", - " uncertainty_correction_threshold=(0.9, 0.995), # randomly sample between 90% and 99.5% coverage required in a month\n", + " uncertainty_correction_threshold=(\n", + " 0.9,\n", + " 0.995,\n", + " ), # randomly sample between 90% and 99.5% coverage required in a month\n", ")" ] }, diff --git a/examples/05_eya_gap_analysis.ipynb b/examples/05_eya_gap_analysis.ipynb index 2e2b4e107..edd9b24b1 100644 --- a/examples/05_eya_gap_analysis.ipynb +++ b/examples/05_eya_gap_analysis.ipynb @@ -47,7 +47,7 @@ "outputs": [], "source": [ "# Load plant object and process plant data\n", - "project = project_ENGIE.prepare('./data/la_haute_borne', use_cleansed=False)\n", + "project = project_ENGIE.prepare(\"./data/la_haute_borne\", use_cleansed=False)\n", "\n", "# Add the analysis workflow validations needed below and re-validate\n", "project.analysis_type.extend([\"TurbineLongTermGrossEnergy\", \"ElectricalLosses\"])\n", @@ -78,8 +78,8 @@ ], "source": [ "# Calculate AEP\n", - "pa = project.MonteCarloAEP(reanalysis_products = ['era5', 'merra2'])\n", - "pa.run(num_sim=20000, reanalysis_products=['era5', 'merra2'])" + "pa = project.MonteCarloAEP(reanalysis_products=[\"era5\", \"merra2\"])\n", + "pa.run(num_sim=20000, reanalysis_products=[\"era5\", \"merra2\"])" ] }, { @@ -100,11 +100,11 @@ "ta = project.TurbineLongTermGrossEnergy(\n", " UQ=True,\n", " num_sim=75,\n", - " max_power_filter=(0.8, 0.9), \n", + " max_power_filter=(0.8, 0.9),\n", " wind_bin_threshold=(1.0, 3.0),\n", " correction_threshold=(0.85, 0.95),\n", ")\n", - "ta.run(reanalysis_products=['era5', 'merra2'])" + "ta.run(reanalysis_products=[\"era5\", \"merra2\"])" ] }, { @@ -161,7 +161,7 @@ "aep = pa.results.aep_GWh.mean()\n", "avail = pa.results.avail_pct.mean()\n", "elec = el.electrical_losses[0][0]\n", - "tie = ta.plant_gross[0][0]/1e6\n", + "tie = ta.plant_gross[0][0] / 1e6\n", "\n", "print(f\"AEP = {aep:21.2f} GWh/yr\")\n", "print(f\"Availability Losses = {avail:.2%}\")\n", @@ -180,7 +180,7 @@ " aep=aep, # AEP (GWh/yr)\n", " availability_losses=avail, # Availability loss (fraction)\n", " electrical_losses=elec, # Electrical loss (fraction)\n", - " turbine_ideal_energy=tie # Turbine ideal energy (GWh/yr)\n", + " turbine_ideal_energy=tie, # Turbine ideal energy (GWh/yr)\n", ")\n", "\n", "# Define EYA data (we are fabricating these data as an example)\n", diff --git a/examples/06_wake_loss_analysis.ipynb b/examples/06_wake_loss_analysis.ipynb index 696769979..9f864a00d 100644 --- a/examples/06_wake_loss_analysis.ipynb +++ b/examples/06_wake_loss_analysis.ipynb @@ -1756,6 +1756,7 @@ "from bokeh.plotting import show\n", "from bokeh.io import output_notebook\n", "from bokeh.resources import INLINE\n", + "\n", "output_notebook(INLINE)\n", "\n", "from openoa.analysis.wake_losses import WakeLosses\n", @@ -1850,7 +1851,7 @@ } ], "source": [ - "show(plot.plot_windfarm(project.asset,tile_name=\"OpenMap\",plot_width=600,plot_height=600))" + "show(plot.plot_windfarm(project.asset, tile_name=\"OpenMap\", plot_width=600, plot_height=600))" ] }, { @@ -1868,7 +1869,7 @@ "metadata": {}, "outputs": [], "source": [ - "# Modify the SCADA data frame so we can access columns using the (variable, turbine ID) \n", + "# Modify the SCADA data frame so we can access columns using the (variable, turbine ID)\n", "# pair as the column name\n", "scada_df = project.scada.unstack()" ] @@ -2009,12 +2010,17 @@ "source": [ "for ind1, tid1 in enumerate(project.turbine_ids):\n", " for tid2 in [tid for ind, tid in enumerate(project.turbine_ids) if ind > ind1]:\n", - " \n", " # limit to time steps when both turbines are producing power\n", - " valid_indices = (scada_df[(\"WTUR_W\",tid1)] >= 0) & (scada_df[(\"WTUR_W\",tid2)] >= 0)\n", + " valid_indices = (scada_df[(\"WTUR_W\", tid1)] >= 0) & (scada_df[(\"WTUR_W\", tid2)] >= 0)\n", "\n", " plt.figure(figsize=(12, 8))\n", - " plt.plot(scada_df.index[valid_indices],met.wrap_180(scada_df.loc[valid_indices,(\"WMET_HorWdDir\",tid2)].values - scada_df.loc[valid_indices,(\"WMET_HorWdDir\",tid1)].values))\n", + " plt.plot(\n", + " scada_df.index[valid_indices],\n", + " met.wrap_180(\n", + " scada_df.loc[valid_indices, (\"WMET_HorWdDir\", tid2)].values\n", + " - scada_df.loc[valid_indices, (\"WMET_HorWdDir\", tid1)].values\n", + " ),\n", + " )\n", " plt.xlabel(\"Time\")\n", " plt.ylabel(\"Wind Direction Difference (deg)\")\n", " plt.title(f\"{tid2} relative to {tid1}\")" @@ -2115,15 +2121,19 @@ "\n", "for tid1 in turbine_ids_sub:\n", " for tid2 in turbine_ids_sub:\n", - " \n", " # limit to time steps when both turbines are producing power\n", - " valid_indices = (scada_df[(\"WTUR_W\",tid1)] >= 0) & (scada_df[(\"WTUR_W\",tid2)] >= 0)\n", - " \n", + " valid_indices = (scada_df[(\"WTUR_W\", tid1)] >= 0) & (scada_df[(\"WTUR_W\", tid2)] >= 0)\n", + "\n", " # further limit to dates prior to 11/25/2015\n", " valid_indices = valid_indices & (scada_df.index < \"2015-11-25 00:00\")\n", - " \n", - " wind_direction_bias_df.loc[tid1,tid2] = np.mean(met.wrap_180(scada_df.loc[valid_indices, (\"WMET_HorWdDir\",tid2)].values - scada_df.loc[valid_indices, (\"WMET_HorWdDir\",tid1)].values))\n", - " \n", + "\n", + " wind_direction_bias_df.loc[tid1, tid2] = np.mean(\n", + " met.wrap_180(\n", + " scada_df.loc[valid_indices, (\"WMET_HorWdDir\", tid2)].values\n", + " - scada_df.loc[valid_indices, (\"WMET_HorWdDir\", tid1)].values\n", + " )\n", + " )\n", + "\n", "wind_direction_bias_df" ] }, @@ -2172,41 +2182,60 @@ } ], "source": [ - "turbine_id_pairs = [(\"R80711\",\"R80790\"),(\"R80736\",\"R80721\")]\n", + "turbine_id_pairs = [(\"R80711\", \"R80790\"), (\"R80736\", \"R80721\")]\n", "\n", "for tid_up, tid_down in turbine_id_pairs:\n", - "\n", " # limit to time steps when both turbines are producing power\n", - " valid_indices = (scada_df[(\"WTUR_W\",tid_up)] >= 0) & (scada_df[(\"WTUR_W\",tid_down)] >= 0)\n", + " valid_indices = (scada_df[(\"WTUR_W\", tid_up)] >= 0) & (scada_df[(\"WTUR_W\", tid_down)] >= 0)\n", "\n", " # further limit to dates prior to 11/25/2015\n", " valid_indices = valid_indices & (scada_df.index < \"2015-11-25 00:00\")\n", "\n", - " scada_df[(\"wind_direction_bin\",tid_up)] = scada_df[(\"WMET_HorWdDir\",tid_up)].round()\n", + " scada_df[(\"wind_direction_bin\", tid_up)] = scada_df[(\"WMET_HorWdDir\", tid_up)].round()\n", "\n", - " scada_df_bin = scada_df.loc[valid_indices].groupby((\"wind_direction_bin\",tid_up)).mean()\n", + " scada_df_bin = scada_df.loc[valid_indices].groupby((\"wind_direction_bin\", tid_up)).mean()\n", "\n", - " turbine_pair_direction = project.turbine_direction_matrix().loc[tid_down,tid_up]\n", + " turbine_pair_direction = project.turbine_direction_matrix().loc[tid_down, tid_up]\n", "\n", " # ratio between mean power of downstream and upstream turbines vs. wind direction\n", - " power_ratio = scada_df_bin[(\"WTUR_W\",tid_down)]/scada_df_bin[(\"WTUR_W\",tid_up)]\n", - " \n", - " # Find direction where peak wake losses are observed, assuming this direction is within 45 degrees \n", + " power_ratio = scada_df_bin[(\"WTUR_W\", tid_down)] / scada_df_bin[(\"WTUR_W\", tid_up)]\n", + "\n", + " # Find direction where peak wake losses are observed, assuming this direction is within 45 degrees\n", " # of actual direction between turbines\n", - " peak_wake_loss_direction = np.round(turbine_pair_direction) - 45.0 + np.argmin(power_ratio[np.round(turbine_pair_direction)-45:np.round(turbine_pair_direction)+45])\n", + " peak_wake_loss_direction = (\n", + " np.round(turbine_pair_direction)\n", + " - 45.0\n", + " + np.argmin(\n", + " power_ratio[\n", + " np.round(turbine_pair_direction) - 45 : np.round(turbine_pair_direction) + 45\n", + " ]\n", + " )\n", + " )\n", "\n", " # Wind direction offset\n", - " direction_offset = np.round(met.wrap_180(peak_wake_loss_direction - turbine_pair_direction),2)\n", - " \n", - " plt.figure(figsize=(9,6))\n", - " plt.plot(power_ratio,label=\"_nolabel_\")\n", - "\n", - " plt.plot(2*[turbine_pair_direction],[power_ratio.min(),power_ratio.max()],'k--',label = \"Direction between Turbines\")\n", - " plt.plot(2*[peak_wake_loss_direction],[power_ratio.min(),power_ratio.max()],'r--',label = \"Direction of Peak Wake Losses\")\n", + " direction_offset = np.round(met.wrap_180(peak_wake_loss_direction - turbine_pair_direction), 2)\n", + "\n", + " plt.figure(figsize=(9, 6))\n", + " plt.plot(power_ratio, label=\"_nolabel_\")\n", + "\n", + " plt.plot(\n", + " 2 * [turbine_pair_direction],\n", + " [power_ratio.min(), power_ratio.max()],\n", + " \"k--\",\n", + " label=\"Direction between Turbines\",\n", + " )\n", + " plt.plot(\n", + " 2 * [peak_wake_loss_direction],\n", + " [power_ratio.min(), power_ratio.max()],\n", + " \"r--\",\n", + " label=\"Direction of Peak Wake Losses\",\n", + " )\n", " plt.legend()\n", " plt.xlabel(\"Wind Direction (deg)\")\n", " plt.ylabel(\"$P_{down}/P_{up}$ (-)\")\n", - " plt.title(f\"Upstream Turbine = {tid_up}, Downstream Turbine = {tid_down}. Wind Direction Offset = {direction_offset} deg.\")" + " plt.title(\n", + " f\"Upstream Turbine = {tid_up}, Downstream Turbine = {tid_down}. Wind Direction Offset = {direction_offset} deg.\"\n", + " )" ] }, { @@ -2257,9 +2286,9 @@ " wind_direction_asset_ids=[\"R80711\", \"R80721\", \"R80736\"],\n", " start_date=None,\n", " end_date=\"2015-11-25 00:00\",\n", - " reanalysis_products=[\"merra2\",\"era5\"],\n", + " reanalysis_products=[\"merra2\", \"era5\"],\n", " end_date_lt=None,\n", - " UQ=False\n", + " UQ=False,\n", ")" ] }, @@ -2316,7 +2345,7 @@ " ws_bin_width_LT_corr=1.0,\n", " num_years_LT=20,\n", " assume_no_wakes_high_ws_LT_corr=True,\n", - " no_wakes_ws_thresh_LT_corr=15.0\n", + " no_wakes_ws_thresh_LT_corr=15.0,\n", ")" ] }, @@ -2452,10 +2481,14 @@ ], "source": [ "for i in range(len(wl.turbine_ids)):\n", - " print(f\"Period-of-record wake losses for turbine {wl.turbine_ids[i]}: {np.round(100*wl.turbine_wake_losses_por[i],2)}%\")\n", - " \n", + " print(\n", + " f\"Period-of-record wake losses for turbine {wl.turbine_ids[i]}: {np.round(100 * wl.turbine_wake_losses_por[i], 2)}%\"\n", + " )\n", + "\n", "for i in range(len(wl.turbine_ids)):\n", - " print(f\"Long-term corrected wake losses for turbine {wl.turbine_ids[i]}: {np.round(100*wl.turbine_wake_losses_lt[i],2)}%\")" + " print(\n", + " f\"Long-term corrected wake losses for turbine {wl.turbine_ids[i]}: {np.round(100 * wl.turbine_wake_losses_lt[i], 2)}%\"\n", + " )" ] }, { @@ -2482,7 +2515,7 @@ } ], "source": [ - "axes = wl.plot_wake_losses_by_wind_direction(turbine_id = \"R80711\")" + "axes = wl.plot_wake_losses_by_wind_direction(turbine_id=\"R80711\")" ] }, { @@ -2509,7 +2542,7 @@ } ], "source": [ - "axes = wl.plot_wake_losses_by_wind_direction(turbine_id = \"R80721\")" + "axes = wl.plot_wake_losses_by_wind_direction(turbine_id=\"R80721\")" ] }, { @@ -2538,7 +2571,7 @@ } ], "source": [ - "axes = wl.plot_wake_losses_by_wind_speed(turbine_id = \"R80711\")" + "axes = wl.plot_wake_losses_by_wind_speed(turbine_id=\"R80711\")" ] }, { @@ -2565,7 +2598,7 @@ } ], "source": [ - "axes = wl.plot_wake_losses_by_wind_speed(turbine_id = \"R80721\")" + "axes = wl.plot_wake_losses_by_wind_speed(turbine_id=\"R80721\")" ] }, { @@ -2597,9 +2630,9 @@ " wind_direction_asset_ids=[\"R80711\", \"R80721\", \"R80736\"],\n", " start_date=None,\n", " end_date=\"2015-11-25 00:00\",\n", - " reanalysis_products=[\"merra2\",\"era5\"],\n", + " reanalysis_products=[\"merra2\", \"era5\"],\n", " end_date_lt=None,\n", - " UQ=True\n", + " UQ=True,\n", ")" ] }, @@ -2641,7 +2674,7 @@ " ws_bin_width_LT_corr=1.0,\n", " num_years_LT=(10, 20),\n", " assume_no_wakes_high_ws_LT_corr=True,\n", - " no_wakes_ws_thresh_LT_corr=15.0\n", + " no_wakes_ws_thresh_LT_corr=15.0,\n", ")" ] }, @@ -2683,8 +2716,12 @@ "print(f\"Standard deviation of period-of-record wake losses: {wl_uq.wake_losses_por_std:.2%}\")\n", "print(f\"Standard deviation of long-term corrected wake losses: {wl_uq.wake_losses_lt_std:.2%}\")\n", "\n", - "print(f\"95% confidence interval of period-of-record wake losses: [{np.percentile(wl_uq.wake_losses_por,2.5):.2%}, {np.percentile(wl_uq.wake_losses_por,97.5):.2%}]\")\n", - "print(f\"95% confidence interval of long-term corrected wake losses: [{np.percentile(wl_uq.wake_losses_por,2.5):.2%}, {np.percentile(wl_uq.wake_losses_lt,97.5):.2%}]\")" + "print(\n", + " f\"95% confidence interval of period-of-record wake losses: [{np.percentile(wl_uq.wake_losses_por, 2.5):.2%}, {np.percentile(wl_uq.wake_losses_por, 97.5):.2%}]\"\n", + ")\n", + "print(\n", + " f\"95% confidence interval of long-term corrected wake losses: [{np.percentile(wl_uq.wake_losses_por, 2.5):.2%}, {np.percentile(wl_uq.wake_losses_lt, 97.5):.2%}]\"\n", + ")" ] }, { @@ -2795,11 +2832,19 @@ "for i in range(len(wl_uq.turbine_ids)):\n", " print(f\"Turbine {wl_uq.turbine_ids[i]}:\")\n", " print(f\"Mean period-of-record wake losses: {wl_uq.turbine_wake_losses_por_mean[i]:.2%}\")\n", - " print(f\"Standard deviation of period-of-record wake losses: {wl_uq.turbine_wake_losses_por_std[i]:.2%}\")\n", - " print(f\"95% confidence interval of period-of-record wake losses: [{np.percentile(wl_uq.turbine_wake_losses_por[:,i],2.5):.2%}, {np.percentile(wl_uq.turbine_wake_losses_por[:,i],97.5):.2%}]\")\n", + " print(\n", + " f\"Standard deviation of period-of-record wake losses: {wl_uq.turbine_wake_losses_por_std[i]:.2%}\"\n", + " )\n", + " print(\n", + " f\"95% confidence interval of period-of-record wake losses: [{np.percentile(wl_uq.turbine_wake_losses_por[:, i], 2.5):.2%}, {np.percentile(wl_uq.turbine_wake_losses_por[:, i], 97.5):.2%}]\"\n", + " )\n", " print(f\"Mean long-term corrected wake losses: {wl_uq.turbine_wake_losses_lt_mean[i]:.2%}\")\n", - " print(f\"Standard deviation of long-term corrected wake losses: {wl_uq.turbine_wake_losses_lt_std[i]:.2%}\")\n", - " print(f\"95% confidence interval of long-term corrected wake losses: [{np.percentile(wl_uq.turbine_wake_losses_lt[:,i],2.5):.2%}, {np.percentile(wl_uq.turbine_wake_losses_lt[:,i],97.5):.2%}]\\n\")" + " print(\n", + " f\"Standard deviation of long-term corrected wake losses: {wl_uq.turbine_wake_losses_lt_std[i]:.2%}\"\n", + " )\n", + " print(\n", + " f\"95% confidence interval of long-term corrected wake losses: [{np.percentile(wl_uq.turbine_wake_losses_lt[:, i], 2.5):.2%}, {np.percentile(wl_uq.turbine_wake_losses_lt[:, i], 97.5):.2%}]\\n\"\n", + " )" ] }, { @@ -2846,7 +2891,7 @@ } ], "source": [ - "axes=wl_uq.plot_wake_losses_by_wind_direction(turbine_id=\"R80721\")" + "axes = wl_uq.plot_wake_losses_by_wind_direction(turbine_id=\"R80721\")" ] }, { @@ -3055,7 +3100,7 @@ "source": [ "plt.figure(figsize=(10, 7))\n", "for id in project.turbine_ids:\n", - " plt.plot(df_speedup[\"wd\"],df_speedup[id],label=id)\n", + " plt.plot(df_speedup[\"wd\"], df_speedup[id], label=id)\n", "plt.xlabel(\"Wind Direction (deg)\")\n", "plt.ylabel(\"Relative Speedup Factor (-)\")\n", "plt.legend()" @@ -3096,11 +3141,11 @@ " wind_direction_asset_ids=[\"R80711\", \"R80721\", \"R80736\"],\n", " start_date=None,\n", " end_date=\"2015-11-25 00:00\",\n", - " reanalysis_products=[\"merra2\",\"era5\"],\n", + " reanalysis_products=[\"merra2\", \"era5\"],\n", " end_date_lt=None,\n", " UQ=False,\n", " correct_for_ws_heterogeneity=True,\n", - " ws_speedup_factor_map=\"example_la_haute_borne_ws_speedup_factors.csv\"\n", + " ws_speedup_factor_map=\"example_la_haute_borne_ws_speedup_factors.csv\",\n", ")" ] }, @@ -3135,7 +3180,7 @@ " ws_bin_width_LT_corr=1.0,\n", " num_years_LT=20,\n", " assume_no_wakes_high_ws_LT_corr=True,\n", - " no_wakes_ws_thresh_LT_corr=15.0\n", + " no_wakes_ws_thresh_LT_corr=15.0,\n", ")" ] }, @@ -3165,8 +3210,12 @@ "source": [ "print(f\"Basic period-of-record wake losses: {wl.wake_losses_por:.2%}\")\n", "print(f\"Basic long-term corrected wake losses: {wl.wake_losses_lt:.2%}\")\n", - "print(f\"Period-of-record wake losses with wind speed heterogeneity corrections: {wl_het.wake_losses_por:.2%}\")\n", - "print(f\"Long-term corrected wake losses with wind speed heterogeneity corrections: {wl_het.wake_losses_lt:.2%}\")" + "print(\n", + " f\"Period-of-record wake losses with wind speed heterogeneity corrections: {wl_het.wake_losses_por:.2%}\"\n", + ")\n", + "print(\n", + " f\"Long-term corrected wake losses with wind speed heterogeneity corrections: {wl_het.wake_losses_lt:.2%}\"\n", + ")" ] }, { @@ -3290,18 +3339,26 @@ ], "source": [ "for i in range(len(wl.turbine_ids)):\n", - " print(f\"Basic period-of-record wake losses for turbine {wl.turbine_ids[i]}: {np.round(100*wl.turbine_wake_losses_por[i],2)}%\")\n", - " \n", + " print(\n", + " f\"Basic period-of-record wake losses for turbine {wl.turbine_ids[i]}: {np.round(100 * wl.turbine_wake_losses_por[i], 2)}%\"\n", + " )\n", + "\n", "for i in range(len(wl.turbine_ids)):\n", - " print(f\"Basic long-term corrected wake losses for turbine {wl.turbine_ids[i]}: {np.round(100*wl.turbine_wake_losses_lt[i],2)}%\")\n", + " print(\n", + " f\"Basic long-term corrected wake losses for turbine {wl.turbine_ids[i]}: {np.round(100 * wl.turbine_wake_losses_lt[i], 2)}%\"\n", + " )\n", "\n", "print(\"\")\n", "\n", "for i in range(len(wl_het.turbine_ids)):\n", - " print(f\"Period-of-record wake losses for turbine {wl_het.turbine_ids[i]} with wind speed heterogeneity corrections: {np.round(100*wl_het.turbine_wake_losses_por[i],2)}%\")\n", - " \n", + " print(\n", + " f\"Period-of-record wake losses for turbine {wl_het.turbine_ids[i]} with wind speed heterogeneity corrections: {np.round(100 * wl_het.turbine_wake_losses_por[i], 2)}%\"\n", + " )\n", + "\n", "for i in range(len(wl_het.turbine_ids)):\n", - " print(f\"Long-term corrected wake losses for turbine {wl_het.turbine_ids[i]} with wind speed heterogeneity corrections: {np.round(100*wl_het.turbine_wake_losses_lt[i],2)}%\")" + " print(\n", + " f\"Long-term corrected wake losses for turbine {wl_het.turbine_ids[i]} with wind speed heterogeneity corrections: {np.round(100 * wl_het.turbine_wake_losses_lt[i], 2)}%\"\n", + " )" ] }, { @@ -3342,8 +3399,8 @@ } ], "source": [ - "fig, axes = wl.plot_wake_losses_by_wind_direction(turbine_id = \"R80711\", return_fig=True)\n", - "axes[0].set_title(axes[0].get_title()+\", using Basic Wake Loss Estimation Method\")" + "fig, axes = wl.plot_wake_losses_by_wind_direction(turbine_id=\"R80711\", return_fig=True)\n", + "axes[0].set_title(axes[0].get_title() + \", using Basic Wake Loss Estimation Method\")" ] }, { @@ -3373,8 +3430,8 @@ } ], "source": [ - "fig, axes = wl_het.plot_wake_losses_by_wind_direction(turbine_id = \"R80711\", return_fig=True)\n", - "axes[0].set_title(axes[0].get_title()+\", with Wind Speed Heterogeneity Corrections\")" + "fig, axes = wl_het.plot_wake_losses_by_wind_direction(turbine_id=\"R80711\", return_fig=True)\n", + "axes[0].set_title(axes[0].get_title() + \", with Wind Speed Heterogeneity Corrections\")" ] }, { @@ -3413,8 +3470,8 @@ } ], "source": [ - "fig, axes = wl.plot_wake_losses_by_wind_direction(turbine_id = \"R80721\", return_fig=True)\n", - "axes[0].set_title(axes[0].get_title()+\", using Basic Wake Loss Estimation Method\")" + "fig, axes = wl.plot_wake_losses_by_wind_direction(turbine_id=\"R80721\", return_fig=True)\n", + "axes[0].set_title(axes[0].get_title() + \", using Basic Wake Loss Estimation Method\")" ] }, { @@ -3444,8 +3501,8 @@ } ], "source": [ - "fig, axes = wl_het.plot_wake_losses_by_wind_direction(turbine_id = \"R80721\", return_fig=True)\n", - "axes[0].set_title(axes[0].get_title()+\", with Wind Speed Heterogeneity Corrections\")" + "fig, axes = wl_het.plot_wake_losses_by_wind_direction(turbine_id=\"R80721\", return_fig=True)\n", + "axes[0].set_title(axes[0].get_title() + \", with Wind Speed Heterogeneity Corrections\")" ] }, { diff --git a/examples/07_static_yaw_misalignment.ipynb b/examples/07_static_yaw_misalignment.ipynb index 035d4537a..96329df70 100644 --- a/examples/07_static_yaw_misalignment.ipynb +++ b/examples/07_static_yaw_misalignment.ipynb @@ -1809,6 +1809,7 @@ "from bokeh.plotting import show\n", "from bokeh.io import output_notebook\n", "from bokeh.resources import INLINE\n", + "\n", "output_notebook(INLINE)\n", "\n", "from openoa.analysis.yaw_misalignment import StaticYawMisalignment\n", @@ -1902,7 +1903,7 @@ } ], "source": [ - "show(plot.plot_windfarm(project.asset,tile_name=\"OpenMap\",plot_width=600,plot_height=600))" + "show(plot.plot_windfarm(project.asset, tile_name=\"OpenMap\", plot_width=600, plot_height=600))" ] }, { @@ -1941,7 +1942,7 @@ " xlabel=\"Wind Speed (m/s)\",\n", " ylabel=\"Blade Pitch Angle (deg.)\",\n", " xlim=(0, 15),\n", - " ylim=(-2,3),\n", + " ylim=(-2, 3),\n", " max_cols=2,\n", " figure_kwargs={\"figsize\": (12, 8)},\n", ")" @@ -2007,14 +2008,13 @@ "power_bin_mad_thresh = 7.0\n", "\n", "for t in project.turbine_ids:\n", - " \n", " # TODO: apply bin power curve filtering\n", - " df_sub = project.scada.loc[(slice(None), t),:]\n", + " df_sub = project.scada.loc[(slice(None), t), :]\n", " df_sub = df_sub.loc[df_sub[\"WROT_BlPthAngVal\"] <= pitch_threshold]\n", - " \n", + "\n", " # Apply power bin filter\n", - " turb_capac=project.asset.loc[t, \"rated_power\"]\n", - " flag_bin=filters.bin_filter(\n", + " turb_capac = project.asset.loc[t, \"rated_power\"]\n", + " flag_bin = filters.bin_filter(\n", " bin_col=df_sub[\"WTUR_W\"],\n", " value_col=df_sub[\"WMET_HorWdSpd\"],\n", " bin_width=0.04 * 0.94 * turb_capac,\n", @@ -2031,7 +2031,7 @@ " power=df_sub[\"WTUR_W\"],\n", " flag=flag_bin,\n", " flag_labels=(\"Outliers\", \"Power Curve\"),\n", - " legend=True\n", + " legend=True,\n", " )" ] }, @@ -2057,11 +2057,7 @@ "metadata": {}, "outputs": [], "source": [ - "yaw_mis = StaticYawMisalignment(\n", - " plant=project,\n", - " turbine_ids=None,\n", - " UQ=False\n", - ")" + "yaw_mis = StaticYawMisalignment(plant=project, turbine_ids=None, UQ=False)" ] }, { @@ -2106,7 +2102,7 @@ " pitch_thresh=pitch_threshold,\n", " max_power_filter=0.95,\n", " power_bin_mad_thresh=power_bin_mad_thresh,\n", - " use_power_coeff=True\n", + " use_power_coeff=True,\n", ")" ] }, @@ -2137,7 +2133,9 @@ ], "source": [ "for i, t in enumerate(yaw_mis.turbine_ids):\n", - " print(f\"Overall yaw misalignment for Turbine {t}: {np.round(yaw_mis.yaw_misalignment[i],1)} degrees\")" + " print(\n", + " f\"Overall yaw misalignment for Turbine {t}: {np.round(yaw_mis.yaw_misalignment[i], 1)} degrees\"\n", + " )" ] }, { @@ -2205,7 +2203,7 @@ } ], "source": [ - "axes_dict = yaw_mis.plot_yaw_misalignment_by_turbine(return_fig = True)" + "axes_dict = yaw_mis.plot_yaw_misalignment_by_turbine(return_fig=True)" ] }, { @@ -2234,11 +2232,7 @@ "metadata": {}, "outputs": [], "source": [ - "yaw_mis = StaticYawMisalignment(\n", - " plant=project,\n", - " turbine_ids=None,\n", - " UQ=True\n", - ")" + "yaw_mis = StaticYawMisalignment(plant=project, turbine_ids=None, UQ=True)" ] }, { @@ -2280,7 +2274,7 @@ " pitch_thresh=pitch_threshold,\n", " max_power_filter=(0.92, 0.98),\n", " power_bin_mad_thresh=(4.0, 10.0),\n", - " use_power_coeff=True\n", + " use_power_coeff=True,\n", ")" ] }, @@ -2321,9 +2315,15 @@ ], "source": [ "for i, t in enumerate(yaw_mis.turbine_ids):\n", - " print(f\"Mean overall yaw misalignment for Turbine {t}: {np.round(yaw_mis.yaw_misalignment_avg[i],1)} degrees\")\n", - " print(f\"Std. Dev. of overall yaw misalignment for Turbine {t}: {np.round(yaw_mis.yaw_misalignment_std[i],1)} degrees\")\n", - " print(f\"95% confidence interval of overall yaw misalignment for Turbine {t}: [{np.round(yaw_mis.yaw_misalignment_95ci[i,0],1)} deg., {np.round(yaw_mis.yaw_misalignment_95ci[i,1],1)} deg.]\")\n" + " print(\n", + " f\"Mean overall yaw misalignment for Turbine {t}: {np.round(yaw_mis.yaw_misalignment_avg[i], 1)} degrees\"\n", + " )\n", + " print(\n", + " f\"Std. Dev. of overall yaw misalignment for Turbine {t}: {np.round(yaw_mis.yaw_misalignment_std[i], 1)} degrees\"\n", + " )\n", + " print(\n", + " f\"95% confidence interval of overall yaw misalignment for Turbine {t}: [{np.round(yaw_mis.yaw_misalignment_95ci[i, 0], 1)} deg., {np.round(yaw_mis.yaw_misalignment_95ci[i, 1], 1)} deg.]\"\n", + " )" ] }, { @@ -2392,7 +2392,7 @@ } ], "source": [ - "axes_dict = yaw_mis.plot_yaw_misalignment_by_turbine(return_fig = True)" + "axes_dict = yaw_mis.plot_yaw_misalignment_by_turbine(return_fig=True)" ] }, { diff --git a/examples/__init__.py b/examples/__init__.py index 13b909548..0760393b8 100644 --- a/examples/__init__.py +++ b/examples/__init__.py @@ -1,4 +1,5 @@ from pathlib import Path + example_data_path = Path(__file__).parents[0].resolve() / "data" / "la_haute_borne" example_data_path_str = str(example_data_path) diff --git a/examples/project_Cubico.py b/examples/project_Cubico.py index 02d07ea31..10cfcaa7c 100644 --- a/examples/project_Cubico.py +++ b/examples/project_Cubico.py @@ -1,7 +1,6 @@ -""" -This is the import script for Cubico's Kelmarsh & Penmanshiel projects. These projects are +"""This is the import script for Cubico's Kelmarsh & Penmanshiel projects. These projects are available under a Creative Commons Attribution 4.0 International license (CC-BY-4.0), and are cited -below: +below. *Kelmarsh*: @@ -35,7 +34,6 @@ import json from pathlib import Path -from zipfile import ZipFile import yaml import pandas as pd @@ -44,14 +42,14 @@ from openoa.plant import PlantData from openoa.logging import logging + logger = logging.getLogger() def download_asset_data( asset: str = "kelmarsh", outfile_path: str | Path = "data/kelmarsh" ) -> None: - """ - Simplify downloading of known open data assets from Zenodo. + """Simplify downloading of known open data assets from Zenodo. Saves the following files to the outfile_path: 1. "record_details.json", which details the Zenodo API details. @@ -67,8 +65,8 @@ def download_asset_data( Raises: NameError: if `asset` is not kelmarsh or penmanshiel. - """ + """ if asset.lower() == "kelmarsh": record_id = 7212475 elif asset.lower() == "penmanshiel": @@ -80,16 +78,15 @@ def download_asset_data( def get_scada_headers(scada_files: list[str]) -> pd.DataFrame: - """ - Get just the headers from the SCADA files. + """Get just the headers from the SCADA files. Args: scada_files(obj:`list[str]`): List of SCADA file paths. Returns: scada_headers(:obj:`dataframe`): Dataframe containing details of all the SCADA files. - """ + """ csv_params = { "index_col": 0, "skiprows": 2, @@ -111,18 +108,17 @@ def get_scada_headers(scada_files: list[str]) -> pd.DataFrame: def get_scada_df(scada_headers: pd.DataFrame, use_columns: list[str] | None = None) -> pd.DataFrame: - """ - Extract the desired SCADA data. + """Extract the desired SCADA data. Args: scada_headers(:obj:`dataframe`): Dataframe containing details of all SCADA files. - usecolumns(obj:`list[str]`): Selection of columns to be imported from the SCADA files. + use_columns(obj:`list[str]`): Selection of columns to be imported from the SCADA files. Defaults to None. Returns: scada(:obj:`dataframe`): Dataframe with SCADA data. - """ + """ if use_columns is None: use_columns = [ "# Date and time", @@ -141,7 +137,7 @@ def get_scada_df(scada_headers: pd.DataFrame, use_columns: list[str] | None = No "usecols": use_columns, } - scada_lst = list() + scada_lst = [] for turbine in scada_headers["Turbine"].unique(): scada_wt = pd.concat( pd.read_csv(f, **csv_params) @@ -158,16 +154,15 @@ def get_scada_df(scada_headers: pd.DataFrame, use_columns: list[str] | None = No def get_curtailment_df(scada_headers: pd.DataFrame) -> pd.DataFrame: - """ - Get the curtailment and availability data. + """Get the curtailment and availability data. Args: scada_headers(:obj:`dataframe`): Dataframe containing details of all SCADA files. Returns: curtailment_df(:obj:`dataframe`): Dataframe with curtailment data. - """ + """ # Curtailment data is available as a subset of the SCADA data use_columns = [ "# Date and time", @@ -181,16 +176,15 @@ def get_curtailment_df(scada_headers: pd.DataFrame) -> pd.DataFrame: def get_meter_data(path: str = "data/kelmarsh") -> pd.DataFrame: - """ - Get the PMU meter data. + """Get the PMU meter data. Args: path(:obj:`str`): Path to meter data. Defaults to "data/kelmarsh". Returns: meter_df(:obj:`dataframe`): Dataframe with meter data. - """ + """ use_columns = ["# Date and time", "GMS Energy Export (kWh)"] csv_params = { @@ -210,8 +204,7 @@ def get_meter_data(path: str = "data/kelmarsh") -> pd.DataFrame: def prepare(asset: str = "kelmarsh", return_value: str = "plantdata") -> PlantData | pd.DataFrame: - """ - Do all loading and preparation of the data for this plant. + """Do all loading and preparation of the data for this plant. Args: asset(:obj:`str`): Asset name, currently either "kelmarsh" or "penmanshiel". Defaults @@ -224,8 +217,8 @@ def prepare(asset: str = "kelmarsh", return_value: str = "plantdata") -> PlantDa Returns: Either PlantData object or Dataframes dependent upon return_value. - """ + """ # Set the path to store and access all the data path = f"data/{asset}" @@ -278,20 +271,20 @@ def prepare(asset: str = "kelmarsh", return_value: str = "plantdata") -> PlantDa logger.info("Reading in the reanalysis data") # reanalysis datasets are held in a dictionary - reanalysis_dict = dict() + reanalysis_dict = {} # MERRA2 from Zenodo asset_path = Path(path).resolve() if (asset_path / f"{asset}_merra2.csv").exists(): logger.info("Reading MERRA2") reanalysis_merra2_df = pd.read_csv(f"{path}/{asset}_merra2.csv") - reanalysis_dict.update(dict(merra2=reanalysis_merra2_df)) + reanalysis_dict.update({"merra2": reanalysis_merra2_df}) # ERA5 from Zenodo if (asset_path / f"{asset}_era5.csv").exists(): logger.info("Reading ERA5") reanalysis_era5_df = pd.read_csv(f"{path}/{asset}_era5.csv") - reanalysis_dict.update(dict(era5=reanalysis_era5_df)) + reanalysis_dict.update({"era5": reanalysis_era5_df}) # ERA5 monthly 10m from CDS if Path(f"{path}/era5_monthly_10m/{asset}_era5_monthly_10m.csv").exists(): @@ -312,7 +305,7 @@ def prepare(asset: str = "kelmarsh", return_value: str = "plantdata") -> PlantDa reanalysis_era5_monthly_df = pd.read_csv( f"{path}/era5_monthly_10m/{asset}_era5_monthly_10m.csv" ) - reanalysis_dict.update(dict(era5_monthly=reanalysis_era5_monthly_df)) + reanalysis_dict.update({"era5_monthly": reanalysis_era5_monthly_df}) # MERRA2 monthly 10m from GES DISC if Path(f"{path}/merra2_monthly_10m/{asset}_merra2_monthly_10m.csv").exists(): @@ -333,7 +326,7 @@ def prepare(asset: str = "kelmarsh", return_value: str = "plantdata") -> PlantDa reanalysis_merra2_monthly_df = pd.read_csv( f"{path}/merra2_monthly_10m/{asset}_merra2_monthly_10m.csv" ) - reanalysis_dict.update(dict(merra2_monthly=reanalysis_merra2_monthly_df)) + reanalysis_dict.update({"merra2_monthly": reanalysis_merra2_monthly_df}) ################### # PLANT DATA # @@ -408,10 +401,10 @@ def prepare(asset: str = "kelmarsh", return_value: str = "plantdata") -> PlantDa }, } - with open(f"{path}/plant_meta.json", "w") as outfile: + with Path(f"{path}/plant_meta.json").open("w") as outfile: json.dump(asset_json, outfile, indent=2) - with open(f"{path}/plant_meta.yml", "w") as outfile: + with Path(f"{path}/plant_meta.yml").open("w") as outfile: yaml.dump(asset_json, outfile, default_flow_style=False) # Return the appropriate data format diff --git a/examples/project_ENGIE.py b/examples/project_ENGIE.py index d0a0e99b8..1df01b67a 100644 --- a/examples/project_ENGIE.py +++ b/examples/project_ENGIE.py @@ -1,8 +1,7 @@ ################################################# # Data import script for La Haute Borne Project # ################################################# -""" -This is the import script for the example ENGIE La Haute Borne project. Below +"""This is the import script for the example ENGIE La Haute Borne project. Below is a description of data quality for each data frame and an overview of the steps taken to correct the raw data for use in the PRUF OA code. @@ -33,7 +32,6 @@ from __future__ import annotations -import re from pathlib import Path from zipfile import ZipFile @@ -43,16 +41,15 @@ import openoa.utils.unit_conversion as un import openoa.utils.met_data_processing as met from openoa.plant import PlantData -from openoa.utils import filters, timeseries +from openoa.utils import filters from openoa.logging import logging + logger = logging.getLogger() def extract_data(path="data/la_haute_borne"): - """ - Extract zip file containing project engie data. - """ + """Extract zip file containing project ENGIE data.""" path = Path(path).resolve() if not path.exists(): logger.info("Extracting compressed data files") @@ -61,13 +58,14 @@ def extract_data(path="data/la_haute_borne"): def clean_scada(scada_file: str | Path) -> pd.DataFrame: - """Reads in and cleans up the SCADA data + """Reads in and cleans up the SCADA data. Args: scada_file (:obj: `str` | `Path`): The file object corresponding to the SCADA data. Returns: pd.DataFrame: The cleaned up SCADA data that is ready for loading into a `PlantData` object. + """ scada_freq = "10min" @@ -124,6 +122,7 @@ def load_cleansed_data(path: str | Path, return_value="plantdata") -> PlantData: Returns: PlantData | tuple[pandas.DataFrame, ...] + """ logger.info("Reading in the previously cleansed data") @@ -132,10 +131,10 @@ def load_cleansed_data(path: str | Path, return_value="plantdata") -> PlantData: meter_df = pd.read_csv(path / "meter.csv") curtail_df = pd.read_csv(path / "curtail.csv") asset_df = pd.read_csv(path / "asset.csv") - reanalysis = dict( - era5=pd.read_csv(path / "reanalysis_era5.csv"), - merra2=pd.read_csv(path / "reanalysis_merra2.csv"), - ) + reanalysis = { + "era5": pd.read_csv(path / "reanalysis_era5.csv"), + "merra2": pd.read_csv(path / "reanalysis_merra2.csv"), + } # Return the appropriate data format if return_value == "dataframes": @@ -157,17 +156,19 @@ def load_cleansed_data(path: str | Path, return_value="plantdata") -> PlantData: def prepare( - path: str | Path = "data/la_haute_borne", return_value="plantdata", use_cleansed: bool = False + path: str | Path = "data/la_haute_borne", + return_value="plantdata", + *, + use_cleansed: bool = False, ): - """ - Do all loading and preparation of the data for this plant. - args: - - path (str): Path to la_haute_borne data folder. If it doesn't exist, we will try to extract a zip file of the same name. - - scada_df (pandas.DataFrame): Override the scada dataframe with one provided by the user. - - return_value (str): "plantdata" will return a fully constructed PlantData object. "dataframes" will return a list of dataframes instead. - - use_cleansed (bool): Use previously prepared data if the the "cleansed" folder exists above the main `path`. Defaults to False. - """ + """Do all loading and preparation of the data for this plant. + Args: + path (str): Path to la_haute_borne data folder. If it doesn't exist, we will try to extract a zip file of the same name. + return_value (str): "plantdata" will return a fully constructed PlantData object. "dataframes" will return a list of dataframes instead. + use_cleansed (bool): Use previously prepared data if the the "cleansed" folder exists above the main `path`. Defaults to False. + + """ if isinstance(path, str): path = Path(path).resolve() @@ -283,7 +284,7 @@ def prepare( meter_df, curtail_df, asset_df, - dict(era5=reanalysis_era5_df, merra2=reanalysis_merra2_df), + {"era5": reanalysis_era5_df, "merra2": reanalysis_merra2_df}, ) elif return_value == "plantdata": # Build and return PlantData diff --git a/openoa/__init__.py b/openoa/__init__.py index b171099dd..89829989f 100644 --- a/openoa/__init__.py +++ b/openoa/__init__.py @@ -1,12 +1,11 @@ -__version__ = "3.2" - -""" -When bumping version, please be sure to also update parameters in sphinx/conf.py -""" +# NOTE: When bumping version, please be sure to also update parameters in sphinx/conf.py. from openoa.plant import PlantData +__version__ = "3.2" + + def __attach_methods(): from openoa.analysis.aep import create_MonteCarloAEP from openoa.analysis.wake_losses import create_WakeLosses @@ -15,12 +14,12 @@ def __attach_methods(): from openoa.analysis.electrical_losses import create_ElectricalLosses from openoa.analysis.turbine_long_term_gross_energy import create_TurbineLongTermGrossEnergy - setattr(PlantData, "MonteCarloAEP", create_MonteCarloAEP) - setattr(PlantData, "WakeLosses", create_WakeLosses) - setattr(PlantData, "EYAGapAnalysis", create_EYAGapAnalysis) - setattr(PlantData, "ElectricalLosses", create_ElectricalLosses) - setattr(PlantData, "StaticYawMisalignment", create_StaticYawMisalignment) - setattr(PlantData, "TurbineLongTermGrossEnergy", create_TurbineLongTermGrossEnergy) + PlantData.MonteCarloAEP = create_MonteCarloAEP + PlantData.WakeLosses = create_WakeLosses + PlantData.EYAGapAnalysis = create_EYAGapAnalysis + PlantData.ElectricalLosses = create_ElectricalLosses + PlantData.StaticYawMisalignment = create_StaticYawMisalignment + PlantData.TurbineLongTermGrossEnergy = create_TurbineLongTermGrossEnergy __attach_methods() diff --git a/openoa/analysis/_analysis_validators.py b/openoa/analysis/_analysis_validators.py index c7ed3f0e7..7dbcd326a 100644 --- a/openoa/analysis/_analysis_validators.py +++ b/openoa/analysis/_analysis_validators.py @@ -11,6 +11,7 @@ def validate_UQ_input(cls, attribute: attrs.Attribute, value: float | tuple) -> when :py:attr:`UQ` is False. Args: + cls: An OpenOA analysis class object. attribute (attrs.Attribute): The attrs Attribute information for the class attribute being validated. value (float | tuple): The user input to the class attribute. @@ -21,6 +22,7 @@ def validate_UQ_input(cls, attribute: attrs.Attribute, value: float | tuple) -> - If UQ is True, and value is not a length-2 tuple - If UQ is True, and each value is not a float - If UQ is False, and the value is not a float. + """ if cls.UQ: if not isinstance(value, tuple): @@ -46,6 +48,7 @@ def validate_half_closed_0_1_right(cls, attribute: attrs.Attribute, value: float """Validates that the value, or tuple of values is in the half-closed range of (0, 1]. Args: + cls: An OpenOA analysis class object. attribute (attrs.Attribute): The attrs Attribute information for the class attribute being validated. value (float | tuple): The user input to the class attribute. @@ -53,6 +56,7 @@ def validate_half_closed_0_1_right(cls, attribute: attrs.Attribute, value: float Raises: ValueError: Raised if a single input is passed and outside the range of (0, 1]. ValueError: Raised if any of the inputs in the input tuple are outside the range of (0, 1]. + """ if isinstance(value, float): if not 0.0 < value <= 1.0: @@ -70,6 +74,7 @@ def validate_half_closed_0_1_left(cls, attribute: attrs.Attribute, value: float """Validates that the value, or tuple of values is in the half-closed range of [0, 1). Args: + cls: An OpenOA analysis class object. attribute (attrs.Attribute): The attrs Attribute information for the class attribute being validated. value (float | tuple): The user input to the class attribute. @@ -77,6 +82,7 @@ def validate_half_closed_0_1_left(cls, attribute: attrs.Attribute, value: float Raises: ValueError: Raised if a single input is passed and outside the range of [0, 1). ValueError: Raised if any of the inputs in the input tuple are outside the range of [0, 1). + """ if isinstance(value, float): if not 0.0 <= value < 1.0: @@ -97,6 +103,7 @@ def validate_reanalysis_selections( ``PlantData`` object's available reanalyis products are provided. Args: + cls: An OpenOA analysis class object. attribute (attrs.Attribute): The attribute data for :py:attr:`value`. value (list[str] | None): The user-provided values to the class attribute. @@ -104,6 +111,7 @@ def validate_reanalysis_selections( ValueError: Raised if "prodcut" is used in :py:attr:`reanalysis_products`. ValueError: Raised if a reanalysis product key that doesn't exist in the base ``PlantData`` object is provided. + """ valid = [*cls.plant.reanalysis] if None in value or value is None: diff --git a/openoa/analysis/aep.py b/openoa/analysis/aep.py index d6e3ab41e..fe4f227eb 100644 --- a/openoa/analysis/aep.py +++ b/openoa/analysis/aep.py @@ -1,6 +1,7 @@ +"""Provides the :py:class:`MonteCarloAEP` analysis class.""" + from __future__ import annotations -import sys import random import datetime from copy import deepcopy @@ -12,23 +13,27 @@ import statsmodels.api as sm import matplotlib.pyplot as plt from attrs import field, define -from tqdm.auto import tqdm, trange +from tqdm.auto import trange from sklearn.metrics import r2_score, mean_squared_error from matplotlib.markers import MarkerStyle from sklearn.linear_model import LinearRegression from sklearn.model_selection import KFold from openoa.plant import PlantData, convert_to_list -from openoa.utils import plot, filters -from openoa.utils import timeseries as tm -from openoa.utils import unit_conversion as un -from openoa.utils import met_data_processing as mt +from openoa.utils import ( + plot, + filters, + timeseries as tm, + unit_conversion as un, + met_data_processing as mt, +) from openoa.schema import FromDictMixin, ResetValuesMixin from openoa.logging import logging, logged_method_call from openoa.schema.metadata import convert_frequency from openoa.utils.machine_learning_setup import MachineLearningSetup from openoa.analysis._analysis_validators import validate_reanalysis_selections + logger = logging.getLogger(__name__) NDArrayFloat = npt.NDArray[np.float64] @@ -38,8 +43,7 @@ def get_annual_values(data): - """ - This function returns annual summations of values in a pandas Series (or each column of a pandas DataFrame) with a + """This function returns annual summations of values in a pandas Series (or each column of a pandas DataFrame) with a DatetimeIndex index starting from the first row. The purpose of the function is to correctly resample to annual values when the first index does not fall on the beginning of the month. @@ -48,8 +52,8 @@ def get_annual_values(data): Returns: :obj:`numpy.ndarray`: Array containing annual summations for each column of the input data. - """ + """ # shift time index to beginning of first month so resampling by 'MS' groups the data into full years # starting from the beginning of the time series ix_start = data.index[0] @@ -63,8 +67,7 @@ def get_annual_values(data): # TODO: Create an analysis result class that could be used for better results aggregation @define(auto_attribs=True) class MonteCarloAEP(FromDictMixin, ResetValuesMixin): - """ - A serial (Pandas-driven) implementation of the benchmark PRUF operational + """A serial (Pandas-driven) implementation of the benchmark PRUF operational analysis implementation. This module collects standard processing and analysis methods for estimating plant level operational AEP and uncertainty. @@ -123,6 +126,7 @@ class MonteCarloAEP(FromDictMixin, ResetValuesMixin): the IAV adjustment is useful for comparing against short-term estimates of energy production, whereas the exclusion of the IAV is useful for comparing against long-term energy production estimates. Defaults to ``True``. + """ plant: PlantData = field(converter=deepcopy, validator=attrs.validators.instance_of(PlantData)) @@ -230,10 +234,7 @@ class MonteCarloAEP(FromDictMixin, ResetValuesMixin): @logged_method_call def __attrs_post_init__(self): - """ - Initialize the Monte Carlo AEP analysis with data and parameters. - """ - + """Initialize the Monte Carlo AEP analysis with data and parameters.""" if self.reg_temperature and self.reg_wind_direction: analysis_type = "MonteCarloAEP-temp-wd" self.reanalysis_vars.extend(["WMETR_EnvTmp", "WMETR_HorWdSpdU", "WMETR_HorWdSpdV"]) @@ -285,30 +286,30 @@ def __attrs_post_init__(self): def run( self, num_sim: int, - reg_model: str = None, - reanalysis_products: list[str] = None, - uncertainty_meter: float = None, - uncertainty_losses: float = None, - uncertainty_windiness: float | tuple[float, float] = None, - uncertainty_loss_max: float | tuple[float, float] = None, - outlier_detection: bool = None, - uncertainty_outlier: float | tuple[float, float] = None, - uncertainty_nan_energy: float = None, - time_resolution: str = None, + reg_model: str | None = None, + reanalysis_products: list[str] | None = None, + uncertainty_meter: float | None = None, + uncertainty_losses: float | None = None, + uncertainty_windiness: float | tuple[float, float] | None = None, + uncertainty_loss_max: float | tuple[float, float] | None = None, + uncertainty_outlier: float | tuple[float, float] | None = None, + uncertainty_nan_energy: float | None = None, + time_resolution: str | None = None, end_date_lt: str | pd.Timestamp | None = None, - ml_setup_kwargs: dict = None, + ml_setup_kwargs: dict | None = None, + *, + outlier_detection: bool | None = None, progress_bar: bool = True, ) -> None: - """ - Process all appropriate data and run the MonteCarlo AEP analysis. + """Process all appropriate data and run the MonteCarlo AEP analysis. .. note:: If None is provided to any of the inputs, then the last used input value will be used for the analysis, and if no prior values were set, then this is the model's defaults. Args: num_sim(:obj:`int`): number of simulations to perform - reanal_products(obj:`list[str]`) : List of reanalysis products to use for Monte Carlo - sampling. Defaults to None, which pulls all the products contained in + reanalysis_products(obj:`list[str]`) : List of reanalysis products to use for Monte + Carlo sampling. Defaults to None, which pulls all the products contained in :py:attr:`plant.reanalysis`. uncertainty_meter(:obj:`float`): Uncertainty on revenue meter data. Defaults to 0.005. uncertainty_losses(:obj:`float`): Uncertainty on long-term losses. Defaults to 0.05. @@ -342,6 +343,7 @@ def run( Returns: None + """ self.num_sim = num_sim initial_parameters = {} @@ -383,15 +385,15 @@ def run( self.ml_setup_kwargs = ml_setup_kwargs # Write parameters of run to the log file - logged_params = dict( - uncertainty_meter=self.uncertainty_meter, - uncertainty_losses=self.uncertainty_losses, - uncertainty_loss_max=self.uncertainty_loss_max, - uncertainty_windiness=self.uncertainty_windiness, - uncertainty_nan_energy=self.uncertainty_nan_energy, - num_sim=self.num_sim, - reanalysis_products=self.reanalysis_products, - ) + logged_params = { + "uncertainty_meter": self.uncertainty_meter, + "uncertainty_losses": self.uncertainty_losses, + "uncertainty_loss_max": self.uncertainty_loss_max, + "uncertainty_windiness": self.uncertainty_windiness, + "uncertainty_nan_energy": self.uncertainty_nan_energy, + "num_sim": self.num_sim, + "reanalysis_products": self.reanalysis_products, + } logger.info(f"Running with parameters: {logged_params}") # Start the computation @@ -407,16 +409,15 @@ def run( @logged_method_call def groupby_time_res(self, df): - """ - Group pandas dataframe based on the time resolution chosen in the calculation. + """Group pandas dataframe based on the time resolution chosen in the calculation. Args: df(:obj:`dataframe`): dataframe that needs to be grouped based on time resolution used Returns: None - """ + """ if self.time_resolution in ("MS", "ME"): df_grouped = df.groupby(df.index.month).mean() elif self.time_resolution == "D": @@ -428,10 +429,7 @@ def groupby_time_res(self, df): @logged_method_call def calculate_aggregate_dataframe(self): - """ - Perform pre-processing of the plant data to produce a monthly/daily data frame to be used in AEP analysis. - """ - + """Perform pre-processing of the plant data to produce a monthly/daily data frame to be used in AEP analysis.""" # Average to monthly/daily, quantify NaN data self.process_revenue_meter_energy() @@ -448,16 +446,16 @@ def calculate_aggregate_dataframe(self): # Drop any data that have NaN gross energy values or NaN reanalysis data self.aggregate = self.aggregate.dropna( - subset=["gross_energy_gwh"] + [product for product in self.reanalysis_products] + subset=["gross_energy_gwh", *self.reanalysis_products] ) @logged_method_call def process_revenue_meter_energy(self): - """ - Initial creation of monthly data frame: - 1. Populate monthly/daily data frame with energy data summed from 10-min QC'd data - 2. For each monthly/daily value, find percentage of NaN data used in creating it and flag if percentage is - greater than 0 + """Initial creation of monthly data frame. + + 1. Populate monthly/daily data frame with energy data summed from 10-min QC'd data + 2. For each monthly/daily value, find percentage of NaN data used in creating it and flag if percentage is + greater than 0 """ df = self.plant.meter # Get the meter data frame @@ -545,15 +543,14 @@ def process_loss_estimates(self): @logged_method_call def process_reanalysis_data(self): - """ - Process reanalysis data for use in PRUF plant analysis: - - calculate density-corrected wind speed and wind components - - get monthly/daily average wind speeds and components - - calculate monthly/daily average wind direction - - calculate monthly/daily average temperature - - append monthly/daily averages to monthly/daily energy data frame - """ + """Process reanalysis data for use in PRUF plant analysis. + - calculate density-corrected wind speed and wind components + - get monthly/daily average wind speeds and components + - calculate monthly/daily average wind direction + - calculate monthly/daily average temperature + - append monthly/daily averages to monthly/daily energy data frame + """ # Identify start and end dates for long-term correction # First find date range common to all reanalysis products and drop minute field of start date start_date = max( @@ -657,9 +654,7 @@ def process_reanalysis_data(self): @logged_method_call def trim_monthly_df(self): - """ - Remove first and/or last month of data if the raw data had an incomplete number of days. - """ + """Remove first and/or last month of data if the raw data had an incomplete number of days.""" for p in self.aggregate.index[[0, -1]]: # Loop through 1st and last data entry if ( self.aggregate.loc[p, "num_days_expected"] @@ -669,8 +664,7 @@ def trim_monthly_df(self): @logged_method_call def calculate_long_term_losses(self): - """ - This function calculates long-term availability and curtailment losses based on the reported + """This function calculates long-term availability and curtailment losses based on the reported data grouped by the time resolution, filtering for those data that are deemed representative of average plant performance. """ @@ -698,11 +692,9 @@ def calculate_long_term_losses(self): @logged_method_call def setup_monte_carlo_inputs(self): + """Create and populate the data frame defining the simulation parameters. + This data frame is stored as :py:attr:`mc_inputs`. """ - Create and populate the data frame defining the simulation parameters. - This data frame is stored as self.mc_inputs - """ - # Create extra long list of renanalysis product names to sample from reanal_list = list(np.repeat(self.reanalysis_products, self.num_sim)) @@ -732,8 +724,7 @@ def setup_monte_carlo_inputs(self): @logged_method_call def filter_outliers(self, n): - """ - This function filters outliers based on a combination of range filter, unresponsive sensor + """This function filters outliers based on a combination of range filter, unresponsive sensor filter, and window filter. We use a memoized funciton to store the regression data in a dictionary for each @@ -746,8 +737,8 @@ def filter_outliers(self, n): Returns: :obj:`pandas.DataFrame`: Filtered monthly/daily data ready for linear regression - """ + """ reanal = self._run.reanalysis_product # Check if valid data has already been calculated and stored. If so, just return it @@ -860,17 +851,16 @@ def filter_outliers(self, n): @logged_method_call def set_regression_data(self, n): - """ - This will be called for each iteration of the Monte Carlo simulation and will do the following: + """This will be called for each iteration of the Monte Carlo simulation and will do the following. - 1. Randomly sample monthly/daily revenue meter, availabilty, and curtailment data based on specified uncertainties - and correlations - 2. Randomly choose one reanalysis product - 3. Calculate gross energy from randomzied energy data - 4. Normalize gross energy to 30-day months - 5. Filter results to remove months/days with NaN data and with combined losses that exceed the Monte Carlo - sampled max threhold - 6. Return the wind speed and normalized gross energy to be used in the regression relationship + 1. Randomly sample monthly/daily revenue meter, availabilty, and curtailment data based on specified uncertainties + and correlations + 2. Randomly choose one reanalysis product + 3. Calculate gross energy from randomzied energy data + 4. Normalize gross energy to 30-day months + 5. Filter results to remove months/days with NaN data and with combined losses that exceed the Monte Carlo + sampled max threhold + 6. Return the wind speed and normalized gross energy to be used in the regression relationship Args: n(:obj:`int`): The Monte Carlo iteration number @@ -878,6 +868,7 @@ def set_regression_data(self, n): Returns: :obj:`pandas.Series`: Monte-Carlo sampled wind speeds and other variables (temperature, wind direction) if used in the regression :obj:`pandas.Series`: Monte-Carlo sampled normalized gross energy + """ # Get data to use in regression based on filtering result reg_data = self.filter_outliers(n) @@ -915,15 +906,15 @@ def set_regression_data(self, n): @logged_method_call def run_regression(self, n): - """ - Run robust linear regression between Monte-Carlo generated monthly/daily gross energy, - wind speed, temperature and wind direction (if used) + """Run robust linear regression between Monte-Carlo generated monthly/daily gross energy, + wind speed, temperature and wind direction (if used). Args: n(:obj:`int`): The Monte Carlo iteration number. Returns: A trained regression model. + """ reg_data = self.set_regression_data(n) # Get regression data @@ -950,10 +941,7 @@ def run_regression(self, n): # Machine learning models else: ml = MachineLearningSetup(algorithm=self.reg_model, **self.ml_setup_kwargs) - if self.plant.log_level in ("WARNING", "ERROR", "CRITICAL", "INFO"): - verbosity = 0 - else: - verbosity = 2 + verbosity = 0 if self.plant.log_level in ("WARNING", "ERROR", "CRITICAL", "INFO") else 2 # Memoized approach for optimized hyperparameters if self._run.reanalysis_product in self.opt_model: self.opt_model[(self._run.reanalysis_product)].fit( @@ -981,9 +969,8 @@ def run_regression(self, n): return self.opt_model[(self._run.reanalysis_product)] @logged_method_call - def run_AEP_monte_carlo(self, progress_bar: bool = True): - """ - Loop through OA process a number of times and return array of AEP results each time + def run_AEP_monte_carlo(self, *, progress_bar: bool = True): + """Loop through OA process a number of times and return array of AEP results each time. Args: progress_bar(:obj:`bool`): Flag to use a progress bar for the iterations in the AEP @@ -991,8 +978,8 @@ def run_AEP_monte_carlo(self, progress_bar: bool = True): Returns: :obj:`numpy.ndarray` Array of AEP, long-term avail, long-term curtailment calculations - """ + """ num_sim = self.num_sim # Initialize arrays to store metrics and results @@ -1125,15 +1112,15 @@ def run_AEP_monte_carlo(self, progress_bar: bool = True): @logged_method_call def sample_long_term_reanalysis(self): - """ - This function returns the long-term monthly/daily wind speeds based on the Monte-Carlo - generated sample of: + """This function returns the long-term monthly/daily wind speeds based on the Monte-Carlo + generated sample of the following. - 1. The reanalysis product - 2. The number of years to use in the long-term correction + 1. The reanalysis product + 2. The number of years to use in the long-term correction Returns: :obj:`pandas.DataFrame`: the windiness-corrected or 'long-term' monthly/daily wind speeds + """ # Check if valid data has already been calculated and stored. If so, just return it if (self._run.reanalysis_product, self._run.num_years_windiness) in self.long_term_sampling: @@ -1194,8 +1181,7 @@ def sample_long_term_reanalysis(self): @logged_method_call def sample_long_term_losses(self, gross_lt): - """ - This function calculates long-term availability and curtailment losses based on the Monte Carlo sampled + """This function calculates long-term availability and curtailment losses based on the Monte Carlo sampled historical availability and curtailment data. To estimate long-term losses, average percentage monthly losses are weighted by monthly long-term gross energy. @@ -1205,6 +1191,7 @@ def sample_long_term_losses(self, gross_lt): Returns: :obj:`float`: long-term availability loss expressed as fraction :obj:`float`: long-term curtailment loss expressed as fraction + """ mc_avail = self.long_term_losses[0] * self._run.loss_fraction mc_curt = self.long_term_losses[1] * self._run.loss_fraction @@ -1226,11 +1213,12 @@ def plot_normalized_monthly_reanalysis_windspeed( self, xlim: tuple[datetime.datetime, datetime.datetime] = (None, None), ylim: tuple[float, float] = (None, None), - return_fig: bool = False, figure_kwargs: dict | None = None, plot_kwargs: dict | None = None, legend_kwargs: dict | None = None, - ) -> None | tuple[plt.Figure, plt.Axes]: + *, + return_fig: bool = False, + ) -> tuple[plt.Figure, plt.Axes] | None: """Make a plot of the normalized annual average wind speeds from reanalysis data to show general trends for each, and highlighting the period of record for the plant data. @@ -1251,6 +1239,7 @@ def plot_normalized_monthly_reanalysis_windspeed( Returns: None | tuple[matplotlib.pyplot.Figure, matplotlib.pyplot.Axes]: If ``return_fig`` is True, then the figure and axes objects are returned for further tinkering/saving. + """ return plot.plot_monthly_reanalysis_windspeed( data=self.plant.reanalysis, @@ -1269,20 +1258,20 @@ def plot_reanalysis_gross_energy_data( outlier_threshold: int, xlim: tuple[float, float] = (None, None), ylim: tuple[float, float] = (None, None), - return_fig: bool = False, figure_kwargs: dict | None = None, plot_kwargs: dict | None = None, legend_kwargs: dict | None = None, - ) -> None | tuple[plt.Figure, plt.Axes]: - """ - Makes a plot of the gross energy vs wind speed for each reanalysis product, with outliers + *, + return_fig: bool = False, + ) -> tuple[plt.Figure, plt.Axes] | None: + """Makes a plot of the gross energy vs wind speed for each reanalysis product, with outliers highlighted in a contrasting color and separate marker. Args: reanalysis (:obj:`dict[str, pandas.DataFrame]`): :py:attr:`PlantData.reanalysis` dictionary of reanalysis :py:class:`DataFrame`. - outlier_thres (:obj:`float`): outlier threshold (typical range of 1 to 4) which adjusts - outlier sensitivity detection. + outlier_threshold (:obj:`float`): outlier threshold (typical range of 1 to 4) which + adjusts outlier sensitivity detection. xlim (:obj:`tuple[float, float]`, optional): A tuple of datetimes representing the x-axis plotting display limits. Defaults to (None, None). ylim (:obj:`tuple[float, float]`, optional): A tuple of the y-axis plotting display limits. @@ -1298,6 +1287,7 @@ def plot_reanalysis_gross_energy_data( Returns: None | tuple[matplotlib.pyplot.Figure, matplotlib.pyplot.Axes]: If `return_fig` is True, then the figure and axes objects are returned for further tinkering/saving. + """ if figure_kwargs is None: figure_kwargs = {} @@ -1327,7 +1317,7 @@ def plot_reanalysis_gross_energy_data( # Monthly case: apply robust linear regression for outliers detection if self.time_resolution in ("MS", "ME"): - for name, df in self.plant.reanalysis.items(): + for name in self.plant.reanalysis: x = sm.add_constant(valid_aggregate[name]) y = valid_aggregate["gross_energy_gwh"] * 30 / valid_aggregate["num_days_expected"] rlm = sm.RLM(y, x, M=sm.robust.norms.HuberT(t=outlier_threshold)) @@ -1353,7 +1343,7 @@ def plot_reanalysis_gross_energy_data( # Daily/hourly case: apply bin filter for outliers detection else: - for name, df in self.plant.reanalysis.items(): + for name in self.plant.reanalysis: x = valid_aggregate[name] y = valid_aggregate["gross_energy_gwh"] plant_capac = self.plant.metadata.capacity / 1000.0 * self.resample_hours @@ -1397,13 +1387,13 @@ def plot_aggregate_plant_data_timeseries( xlim: tuple[datetime.datetime, datetime.datetime] = (None, None), ylim_energy: tuple[float, float] = (None, None), ylim_loss: tuple[float, float] = (None, None), - return_fig: bool = False, figure_kwargs: dict | None = None, plot_kwargs: dict | None = None, legend_kwargs: dict | None = None, + *, + return_fig: bool = False, ): - """ - Plot timeseries of monthly/daily gross energy, availability and curtailment. + """Plot timeseries of monthly/daily gross energy, availability and curtailment. Args: data(:obj:`pandas.DataFrame`): A pandas DataFrame containing energy production and losses. @@ -1429,6 +1419,7 @@ def plot_aggregate_plant_data_timeseries( None | tuple[matplotlib.pyplot.Figure, tuple[matplotlib.pyplot.Axes, matplotlib.pyplot.Axes]]: If `return_fig` is True, then the figure and axes objects are returned for further tinkering/saving. + """ return plot.plot_plant_energy_losses_timeseries( data=self.aggregate, @@ -1453,13 +1444,13 @@ def plot_result_aep_distributions( ylim_aep: tuple[float, float] = (None, None), ylim_availability: tuple[float, float] = (None, None), ylim_curtail: tuple[float, float] = (None, None), - return_fig: bool = False, figure_kwargs: dict | None = None, plot_kwargs: dict | None = None, annotate_kwargs: dict | None = None, - ) -> None | tuple[plt.Figure, plt.Axes]: - """ - Plot a distribution of AEP values from the Monte-Carlo OA method + *, + return_fig: bool = False, + ) -> tuple[plt.Figure, plt.Axes] | None: + """Plot a distribution of AEP values from the Monte-Carlo OA method. Args: xlim_aep (:obj:`tuple[float, float]`, optional): A tuple of floats representing the x-axis plotting display @@ -1485,6 +1476,7 @@ def plot_result_aep_distributions( Returns: None | tuple[matplotlib.pyplot.Figure, matplotlib.pyplot.Axes]: If `return_fig` is True, then the figure and axes objects are returned for further tinkering/saving. + """ plot_results = self.results.copy() plot_results[["avail_pct", "curt_pct"]] = plot_results[["avail_pct", "curt_pct"]] * 100 @@ -1505,15 +1497,16 @@ def plot_aep_boxplot( x: pd.Series, xlabel: str, ylim: tuple[float, float] = (None, None), - with_points: bool = False, points_label: str = "Individual AEP Estimates", - return_fig: bool = False, figure_kwargs: dict | None = None, plot_kwargs_box: dict | None = None, plot_kwargs_points: dict | None = None, legend_kwargs: dict | None = None, - ) -> None | tuple[plt.Figure, plt.Axes]: - """Plot box plots of AEP results sliced by a specified Monte Carlo parameter + *, + with_points: bool = False, + return_fig: bool = False, + ) -> tuple[plt.Figure, plt.Axes] | None: + """Plot box plots of AEP results sliced by a specified Monte Carlo parameter. Args: x(:obj:`pandas.Series`): The data that splits the results in y. @@ -1538,6 +1531,7 @@ def plot_aep_boxplot( None | tuple[matplotlib.pyplot.Figure, matplotlib.pyplot.Axes, dict]: If `return_fig` is True, then the figure object, axes object, and a dictionary of the boxplot objects are returned for further tinkering/saving. + """ return plot.plot_boxplot( x=x, @@ -1573,23 +1567,24 @@ def plot_aep_boxplot( __defaults_apply_iav = MonteCarloAEP.__attrs_attrs__.apply_iav.default -def create_MonteCarloAEP( +def create_MonteCarloAEP( # ruff: ignore[D103] project: PlantData, reanalysis_products: list[str] = __defaults_reanalysis_products, uncertainty_meter: float = __defaults_uncertainty_meter, uncertainty_losses: float = __defaults_uncertainty_losses, uncertainty_windiness: NDArrayFloat = __defaults_uncertainty_windiness, uncertainty_loss_max: NDArrayFloat = __defaults_uncertainty_loss_max, - outlier_detection: bool = __defaults_outlier_detection, uncertainty_outlier: NDArrayFloat = __defaults_uncertainty_outlier, uncertainty_nan_energy: float = __defaults_uncertainty_nan_energy, time_resolution: str = __defaults_time_resolution, end_date_lt: str | pd.Timestamp = __defaults_end_date_lt, reg_model: str = __defaults_reg_model, ml_setup_kwargs: dict = __defaults_ml_setup_kwargs, + n_jobs: int | None = __defaults_n_jobs, + *, + outlier_detection: bool = __defaults_outlier_detection, reg_temperature: bool = __defaults_reg_temperature, reg_wind_direction: bool = __defaults_reg_wind_direction, - n_jobs: int | None = __defaults_n_jobs, apply_iav: bool = __defaults_apply_iav, ) -> MonteCarloAEP: return MonteCarloAEP( diff --git a/openoa/analysis/electrical_losses.py b/openoa/analysis/electrical_losses.py index e33a80695..72541427b 100644 --- a/openoa/analysis/electrical_losses.py +++ b/openoa/analysis/electrical_losses.py @@ -1,7 +1,8 @@ -# This class defines key analytical routines for calculating electrical losses for -# a wind plant using operational data. Electrical loss is calculated per month and on -# an average annual basis by comparing monthly energy production from the turbines -# and the revenue meter +"""This class defines key analytical routines for calculating electrical losses for +a wind plant using operational data. Electrical loss is calculated per month and on +an average annual basis by comparing monthly energy production from the turbines +and the revenue meter. +""" from __future__ import annotations @@ -23,6 +24,7 @@ from openoa.utils.plot import set_styling from openoa.analysis._analysis_validators import validate_UQ_input, validate_half_closed_0_1_right + logger = logging.getLogger(__name__) set_styling() @@ -34,8 +36,7 @@ @define(auto_attribs=True) class ElectricalLosses(FromDictMixin, ResetValuesMixin): - """ - A serial implementation of calculating the average monthly and annual electrical losses at a + """A serial implementation of calculating the average monthly and annual electrical losses at a wind power plant, and the associated uncertainty. Energy output from the turbine SCADA meter and the wind plant revenue meter are used to estimate electrical losses. @@ -69,6 +70,7 @@ class ElectricalLosses(FromDictMixin, ResetValuesMixin): the range of (0, 1), under which months should be eliminated. If :py:attr:`UQ` = True, then a 2-element tuple containing an upper and lower bound for a randomly selected value should be given, otherwise, a scalar value should be provided. + """ plant: PlantData = field(converter=deepcopy, validator=attrs.validators.instance_of(PlantData)) @@ -104,9 +106,7 @@ class ElectricalLosses(FromDictMixin, ResetValuesMixin): @logged_method_call def __attrs_post_init__(self): - """ - Initialize logging and post-initialization setup steps. - """ + """Initialize logging and post-initialization setup steps.""" if {"ElectricalLosses", "all"}.intersection(self.plant.analysis_type) == set(): self.plant.analysis_type.append("ElectricalLosses") @@ -136,8 +136,7 @@ def run( uncertainty_scada: NDArrayFloat | float = None, uncertainty_correction_threshold: NDArrayFloat | tuple[float, float] | float = None, ): - """ - Run the electrical losses calculation. + """Run the electrical losses calculation. .. note:: If None is provided to any of the inputs, then the last used input value will be used for the analysis, and if no prior values were set, then this is the model's defaults. @@ -152,6 +151,7 @@ def run( the range of (0, 1], under which months should be eliminated. If :py:attr:`UQ` = True, then a 2-element tuple containing an upper and lower bound for a randomly selected value should be given, otherwise, a scalar value should be provided. + """ initial_parameters = {} if num_sim is not None: @@ -183,8 +183,7 @@ def run( @logged_method_call def setup_inputs(self): - """ - Create and populate the data frame defining the simulation parameters. + """Create and populate the data frame defining the simulation parameters. This data frame is stored as self.inputs. """ if self.UQ: @@ -215,8 +214,7 @@ def setup_inputs(self): @logged_method_call def process_scada(self): - """ - Calculate daily sum of turbine energy only for days when all turbines are reporting + """Calculate daily sum of turbine energy only for days when all turbines are reporting at all time steps. """ logger.info("Processing SCADA data") @@ -251,9 +249,7 @@ def process_scada(self): @logged_method_call def process_meter(self): - """ - Calculate daily sum of meter energy only for days when meter data is reporting at all time steps. - """ + """Calculate daily sum of meter energy only for days when meter data is reporting at all time steps.""" logger.info("Processing meter data") meter_df = self.plant.meter.copy() @@ -274,8 +270,7 @@ def process_meter(self): @logged_method_call def calculate_electrical_losses(self): - """ - Apply Monte Carlo approach to calculate electrical losses and their uncertainty based on the + """Apply Monte Carlo approach to calculate electrical losses and their uncertainty based on the difference in the sum of turbine and metered energy over the compiled days. """ logger.info("Calculating electrical losses") @@ -332,11 +327,12 @@ def plot_monthly_losses( self, xlim: tuple[datetime.datetime | None, datetime.datetime | None] = (None, None), ylim: tuple[float | None, float | None] = (None, None), - return_fig: bool = False, figure_kwargs: dict | None = None, legend_kwargs: dict | None = None, plot_kwargs: dict | None = None, - ) -> None | tuple[plt.Figure, plt.Axes]: + *, + return_fig: bool = False, + ) -> tuple[plt.Figure, plt.Axes] | None: """Plots the monthly timeseries of electrical losses as a percent. Args: @@ -356,6 +352,7 @@ def plot_monthly_losses( Returns: None | tuple[plt.Figure, plt.Axes]: If :py:attr:`return_fig`, then return the figure and axes objects in addition to showing the plot. + """ if figure_kwargs is None: figure_kwargs = {} @@ -377,7 +374,7 @@ def plot_monthly_losses( std = losses.std() ax.plot( losses * 100, - label=f"Electrical Losses\n$\\mu$={mean:.2%}, $\\sigma$={std:.2%}", # noqa: W605 + label=f"Electrical Losses\n$\\mu$={mean:.2%}, $\\sigma$={std:.2%}", **plot_kwargs, ) @@ -403,15 +400,16 @@ def plot_monthly_losses( __defaults_uncertainty_scada = ElectricalLosses.__attrs_attrs__.uncertainty_scada.default -def create_ElectricalLosses( +def create_ElectricalLosses( # ruff: ignore[D103] project: PlantData, - UQ: bool = __defaults_UQ, num_sim: int = __defaults_num_sim, uncertainty_correction_threshold: ( NDArrayFloat | tuple[float, float] | float ) = __defaults_uncertainty_correction_threshold, uncertainty_meter: NDArrayFloat | tuple[float, float] | float = __defaults_uncertainty_meter, uncertainty_scada: NDArrayFloat | tuple[float, float] | float = __defaults_uncertainty_scada, + *, + UQ: bool = __defaults_UQ, ) -> ElectricalLosses: return ElectricalLosses( plant=project, diff --git a/openoa/analysis/eya_gap_analysis.py b/openoa/analysis/eya_gap_analysis.py index 10d510b2b..409d6e9c4 100644 --- a/openoa/analysis/eya_gap_analysis.py +++ b/openoa/analysis/eya_gap_analysis.py @@ -1,5 +1,4 @@ -""" -This class defines key analytical routines for performing a 'gap-analysis' on EYA-estimated annual +"""This class defines key analytical routines for performing a 'gap-analysis' on EYA-estimated annual energy production (AEP) and that from operational data. Categories considered are availability, electrical losses, and long-term gross energy. The main output is a 'waterfall' plot linking the EYA- estimated and operational-estimated AEP values. @@ -8,9 +7,6 @@ from __future__ import annotations import attrs -import numpy as np -import pandas as pd -import matplotlib.pyplot as plt from attrs import field, define from openoa.plant import PlantData @@ -19,6 +15,7 @@ from openoa.logging import logging, logged_method_call from openoa.analysis._analysis_validators import validate_half_closed_0_1_left + logger = logging.getLogger(__name__) plot.set_styling() @@ -81,8 +78,7 @@ def validate_0_1(self, attribute: attrs.Attribute, value: float) -> None: @define(auto_attribs=True) class EYAGapAnalysis(FromDictMixin): - """ - Performs a gap analysis between the estimated annual energy production (AEP) from an energy + """Performs a gap analysis between the estimated annual energy production (AEP) from an energy yield estimate (EYA) and the actual AEP as measured from an operational assessment (OA). The gap analysis is based on comparing the following three key metrics: @@ -106,6 +102,7 @@ class EYAGapAnalysis(FromDictMixin): plant(:obj:`PlantData object`): PlantData object from which EYAGapAnalysis should draw data. eya_estimates(:obj:`EYAEstimate`): Numpy array with EYA estimates listed in required order oa_results(:obj:`OAResults`): Numpy array with OA results listed in required order. + """ plant: PlantData = field(validator=attrs.validators.instance_of((PlantData, type(None)))) @@ -128,27 +125,26 @@ def __attrs_post_init__(self): @logged_method_call def run(self): - """ - Run the EYA Gap analysis functions in order by calling this function. + """Run the EYA Gap analysis functions in order by calling this function. Args: (None) Returns: (None) - """ + """ self.compiled_data = self.compile_data() # Compile EYA and OA data logger.info("Gap analysis complete") @logged_method_call def compile_data(self): - """ - Compiles the EYA and OA metrics, and computes the differences. + """Compiles the EYA and OA metrics, and computes the differences. Returns: :obj:`list[float]`: The list of EYA AEP, and differences in turbine gross energy, availability losses, electrical losses, and unaccounted losses. + """ logger.info("Compiling EYA and OA data") @@ -178,22 +174,15 @@ def compile_data(self): def plot_waterfall( self, - index: list[str] = [ - "EYA AEP", - "TIE", - "Availability\nLosses", - "Electrical\nLosses", - "Unexplained", - "OA AEP", - ], + index: list[str] | None = None, ylabel: str = "Energy (GWh/yr)", ylim: tuple[float, float] = (None, None), - return_fig: bool = False, plot_kwargs: dict | None = None, figure_kwargs: dict | None = None, - ) -> None | tuple: - """ - Produce a waterfall plot showing the progression from the EYA estimates to the calculated OA + *, + return_fig: bool = False, + ) -> tuple | None: + """Produce a waterfall plot showing the progression from the EYA estimates to the calculated OA estimates of AEP. Args: @@ -216,7 +205,17 @@ def plot_waterfall( Returns: None | tuple[plt.Figure, plt.Axes]: If :py:attr:`return_fig`, then return the figure and axes objects in addition to showing the plot. + """ + if index is None: + index = [ + "EYA AEP", + "TIE", + "Availability\nLosses", + "Electrical\nLosses", + "Unexplained", + "OA AEP", + ] return plot.plot_waterfall( self.compiled_data, index=index, @@ -228,7 +227,7 @@ def plot_waterfall( ) -def create_EYAGapAnalysis( +def create_EYAGapAnalysis( # ruff: ignore[D103] project: PlantData, eya_estimates: dict | EYAEstimate, oa_results: dict | OAResults ) -> EYAGapAnalysis: return EYAGapAnalysis(project, eya_estimates, oa_results) diff --git a/openoa/analysis/turbine_long_term_gross_energy.py b/openoa/analysis/turbine_long_term_gross_energy.py index 25954d9d3..dcff6206d 100644 --- a/openoa/analysis/turbine_long_term_gross_energy.py +++ b/openoa/analysis/turbine_long_term_gross_energy.py @@ -1,5 +1,4 @@ -""" -This class defines key analytical routines for performing a gap analysis on +"""This class defines key analytical routines for performing a gap analysis on EYA-estimated annual energy production (AEP) and that from operational data. Categories considered are availability, electrical losses, and long-term gross energy. The main output is a 'waterfall' plot linking the EYA-estimated and operational-estiamted AEP values. @@ -9,7 +8,7 @@ import random from copy import deepcopy -from typing import Callable +from collections.abc import Callable import attrs import numpy as np @@ -21,9 +20,7 @@ from matplotlib.ticker import StrMethodFormatter from openoa.plant import PlantData, convert_to_list -from openoa.utils import plot, filters, imputing -from openoa.utils import timeseries as ts -from openoa.utils import met_data_processing as met +from openoa.utils import plot, filters, imputing, timeseries as ts, met_data_processing as met from openoa.schema import FromDictMixin, ResetValuesMixin from openoa.logging import logging, logged_method_call from openoa.utils.power_curve import functions @@ -33,6 +30,7 @@ validate_reanalysis_selections, ) + logger = logging.getLogger(__name__) plot.set_styling() @@ -44,8 +42,7 @@ @define(auto_attribs=True) class TurbineLongTermGrossEnergy(FromDictMixin, ResetValuesMixin): - """ - Calculates long-term gross energy for each turbine in a wind farm using methods implemented in + """Calculates long-term gross energy for each turbine in a wind farm using methods implemented in the utils subpackage for data processing and analysis. The method proceeds as follows: @@ -92,6 +89,7 @@ class TurbineLongTermGrossEnergy(FromDictMixin, ResetValuesMixin): scada energy data should be corrected. When :py:attr:`UQ` is True, then this should be a tuple of the lower and upper limits of this threshold, otherwise a single value should be used. Defaults to (0.85, 0.95) + """ plant: PlantData = field(converter=deepcopy, validator=attrs.validators.instance_of(PlantData)) @@ -151,9 +149,7 @@ class TurbineLongTermGrossEnergy(FromDictMixin, ResetValuesMixin): @logged_method_call def __attrs_post_init__(self): - """ - Runs any non-automated setup steps for the analysis class. - """ + """Runs any non-automated setup steps for the analysis class.""" if {"TurbineLongTermGrossEnergy", "all"}.intersection(self.plant.analysis_type) == set(): self.plant.analysis_type.append("TurbineLongTermGrossEnergy") @@ -188,8 +184,7 @@ def run( max_power_filter: float | tuple[float, float] | None = None, correction_threshold: float | tuple[float, float] | None = None, ) -> None: - """ - Pre-process the run-specific data settings for each simulation, then fit and apply the + """Pre-process the run-specific data settings for each simulation, then fit and apply the model for each simualtion. .. note:: If None is provided to any of the inputs, then the last used input value will be @@ -215,6 +210,7 @@ def run( scada energy data should be corrected. When :py:attr:`UQ` is True, then this should be a tuple of the lower and upper limits of this threshold, otherwise a single value should be used. Defaults to (0.85, 0.95) + """ initial_parameters = {} if num_sim is not None: @@ -261,9 +257,8 @@ def run( self.set_values(initial_parameters) def setup_inputs(self) -> None: - """ - Create and populate the data frame defining the simulation parameters. - This data frame is stored as self._inputs + """Create and populate the data frame defining the simulation parameters. + This data frame is stored as :py:attr:`_inputs`. """ if self.UQ: reanal_list = list( @@ -307,10 +302,7 @@ def setup_inputs(self) -> None: self._inputs = pd.DataFrame(inputs) def sort_scada_by_turbine(self) -> None: - """ - Sorts the SCADA DataFrame by the asset_id and timestamp index columns, respectively. - """ - + """Sorts the SCADA DataFrame by the asset_id and timestamp index columns, respectively.""" df = self.plant.scada.copy() dic = self.scada_dict @@ -324,19 +316,18 @@ def sort_scada_by_turbine(self) -> None: @logged_method_call def filter_turbine_data(self) -> None: - """ - Apply a set of filtering algorithms to the turbine wind speed vs power curve to flag - data not representative of normal turbine operation + """Apply a set of filtering algorithms to the turbine wind speed vs power curve to flag + data not representative of normal turbine operation. Performs the following manipulations: - 1. Drops any scada rows that don't have any windspeed or energy data - 2. Flags windspeed values outside the range [0, 40] - 3. Flags windspeed values that have stayed the same for at least 3 straight readings - 4. Flags power values less than 2% of turbine capacity when wind speed above cut-in - 5. Flags windspeed and power values that don't mutually coincide within a reasonable range - 6. Combine the flags using an "or" combination to be a new column in scada: "flag_final" - """ + 1. Drops any scada rows that don't have any windspeed or energy data + 2. Flags windspeed values outside the range [0, 40] + 3. Flags windspeed values that have stayed the same for at least 3 straight readings + 4. Flags power values less than 2% of turbine capacity when wind speed above cut-in + 5. Flags windspeed and power values that don't mutually coincide within a reasonable range + 6. Combine the flags using an "or" combination to be a new column in scada: "flag_final" + """ dic = self.scada_dict # Loop through turbines @@ -386,9 +377,7 @@ def filter_turbine_data(self) -> None: @logged_method_call def setup_daily_reanalysis_data(self) -> None: - """ - Process reanalysis data to daily means for later use in the GAM model. - """ + """Process reanalysis data to daily means for later use in the GAM model.""" # Memoize the function so you don't have to recompute the same reanalysis product twice if (df_daily := self.reanalysis_memo.get(self._run.reanalysis_product, None)) is not None: self.daily_reanalysis = df_daily.copy() @@ -415,13 +404,10 @@ def setup_daily_reanalysis_data(self) -> None: @logged_method_call def filter_sum_impute_scada(self) -> None: - """ - Filter SCADA data for unflagged data, gather SCADA energy data into daily sums, and correct daily summed + """Filter SCADA data for unflagged data, gather SCADA energy data into daily sums, and correct daily summed energy based on amount of missing data and a threshold limit. Finally impute missing data for each turbine - based on reported energy data from other highly correlated turbines. - threshold + based on reported energy data from other highly correlated turbines threshold. """ - scada = self.scada_dict expected_count = ( HOURS_PER_DAY @@ -504,7 +490,6 @@ def fit_model(self) -> None: """Fit the daily turbine energy sum and atmospheric variable averages using a GAM model using wind speed, wind direction, and air density. """ - mod_dict = self.turbine_model_dict mod_results = self._model_results @@ -526,11 +511,11 @@ def fit_model(self) -> None: @logged_method_call def apply_model(self, i: int) -> None: - """ - Apply the model to the reanalysis data to calculate long-term gross energy for each turbine. + """Apply the model to the reanalysis data to calculate long-term gross energy for each turbine. Args: i(:obj:`int`): The Monte Carlo iteration number. + """ turb_gross = self.turb_lt_gross mod_results = self._model_results @@ -566,15 +551,16 @@ def apply_model(self, i: int) -> None: def plot_filtered_power_curves( self, turbines: list[str] | None = None, - flag_labels: tuple[str, str] = None, + flag_labels: tuple[str, str] | None = None, max_cols: int = 3, xlim: tuple[float, float] = (None, None), ylim: tuple[float, float] = (None, None), - legend: bool = False, - return_fig: bool = False, figure_kwargs: dict | None = None, legend_kwargs: dict | None = None, plot_kwargs: dict | None = None, + *, + legend: bool = False, + return_fig: bool = False, ): """Plot the raw and flagged power curve data. @@ -603,6 +589,7 @@ def plot_filtered_power_curves( Returns: None | tuple[matplotlib.pyplot.Figure, matplotlib.pyplot.Axes]: If `return_fig` is True, then the figure and axes objects are returned for further tinkering/saving. + """ return plot.plot_power_curves( data=self.scada_dict, @@ -628,18 +615,19 @@ def plot_daily_fitting_result( max_cols: int = 3, xlim: tuple[float, float] = (None, None), ylim: tuple[float, float] = (None, None), - legend: bool = False, - return_fig: bool = False, figure_kwargs: dict | None = None, legend_kwargs: dict | None = None, plot_kwargs: dict | None = None, + *, + legend: bool = False, + return_fig: bool = False, ): """Plot the raw, imputed, and modeled power curve data. Args: turbines(:obj:`list[str]`, optional): The list of turbines to be plot, if not all of the keys in :py:attr:`data`. - labels (:obj:`tuple[str, str]`, optional): The labels to give to the scatter points, + flag_labels (:obj:`tuple[str, str]`, optional): The labels to give to the scatter points, corresponding to the modeled, imputed, and input data, respectively. Defaults to ("Modeled", "Imputed", "Input"). max_cols(:obj:`int`, optional): The maximum number of columns in the plot. Defaults to 3. @@ -661,6 +649,7 @@ def plot_daily_fitting_result( Returns: None | tuple[matplotlib.pyplot.Figure, matplotlib.pyplot.Axes]: If :py:attr`return_fig` is True, then the figure and axes objects are returned for further tinkering/saving. + """ if figure_kwargs is None: figure_kwargs = {} @@ -681,7 +670,7 @@ def plot_daily_fitting_result( fig, axes_list = plt.subplots(num_rows, max_cols, **figure_kwargs) ws_daily = self.daily_reanalysis["WMETR_HorWdSpd"] - for i, (t, ax) in enumerate(zip(turbines, axes_list.flatten())): + for i, (t, ax) in enumerate(zip(turbines, axes_list.flatten(), strict=False)): df = self.turbine_model_dict[t] df_imputed = df.loc[df["energy_corrected"] != df["energy_imputed"]] @@ -751,15 +740,16 @@ def plot_daily_fitting_result( ) -def create_TurbineLongTermGrossEnergy( +def create_TurbineLongTermGrossEnergy( # ruff: ignore[D103] project: PlantData, - UQ: bool = __defaults_UQ, num_sim: int = __defaults_num_sim, reanalysis_products=__defaults_reanalysis_products, uncertainty_scada: float = __defaults_uncertainty_scada, wind_bin_threshold: NDArrayFloat = __defaults_wind_bin_threshold, max_power_filter: NDArrayFloat = __defaults_max_power_filter, correction_threshold: NDArrayFloat = __defaults_correction_threshold, + *, + UQ: bool = __defaults_UQ, ) -> TurbineLongTermGrossEnergy: return TurbineLongTermGrossEnergy( plant=project, diff --git a/openoa/analysis/wake_losses.py b/openoa/analysis/wake_losses.py index 4dc32714d..3339b4e77 100644 --- a/openoa/analysis/wake_losses.py +++ b/openoa/analysis/wake_losses.py @@ -1,39 +1,40 @@ -# This class defines key analytical routines for estimating wake losses for an operating -# wind plant using SCADA data. At a high level, for each SCADA time step, freestream wind -# turbines are identified using the turbine coordinates and a reference wind direction -# signal. The mean power production for all turbines in the wind plant is summed over all -# time steps and compared to the mean power of the freestream turbines summed over all time -# steps to estimate wake losses during the period of record. An optional correction can be applied -# to the potential power of the wind plant to account for freestream wind speed heterogeneity -# based on user-provided wind direction-dependent wind speedup factors at each turbine location. -# Methods for calculating the long-term wake losses using reanalaysis data and quantifying -# uncertainty are provided as well. - -# The general approach for estimating wake losses and quantifying uncertainty using bootstrapping -# is based in part on the following publications: -# 1. Barthelmie, R. J. and Jensen, L. E. Evaluation of wind farm efficiency and wind turbine wakes -# at the Nysted offshore wind farm, *Wind Energy* 13(6):573–586 (2010). -# https://doi.org/10.1002/we.408. -# 2. Nygaard, N. G. Systematic quantification of wake model uncertainty. Proc. EWEA Offshore, -# Copenhagen, Denmark, March 10-12 (2015). -# 3. Walker, K., Adams, N., Gribben, B., Gellatly, B., Nygaard, N. G., Henderson, A., Marchante -# Jimémez, M., Schmidt, S. R., Rodriguez Ruiz, J., Paredes, D., Harrington, G., Connell, N., -# Peronne, O., Cordoba, M., Housley, P., Cussons, R., Håkansson, M., Knauer, A., and Maguire, -# E.: An evaluation of the predictive accuracy of wake effects models for offshore wind farms. -# *Wind Energy* 19(5):979–996 (2016). https://doi.org/10.1002/we.1871. -# -# The corrections for freestream wind speed heterogeneity are based in part on the approach -# presented in: -# 4. Kassebaum, J. Wake Validation Through SCADA Data Analysis. Proc. American Clean Power Resource -# & Project Energy Assessment Virtual Summit 2021 (2021). - +"""This class defines key analytical routines for estimating wake losses for an operating +wind plant using SCADA data. At a high level, for each SCADA time step, freestream wind +turbines are identified using the turbine coordinates and a reference wind direction +signal. The mean power production for all turbines in the wind plant is summed over all +time steps and compared to the mean power of the freestream turbines summed over all time +steps to estimate wake losses during the period of record. An optional correction can be applied +to the potential power of the wind plant to account for freestream wind speed heterogeneity +based on user-provided wind direction-dependent wind speedup factors at each turbine location. +Methods for calculating the long-term wake losses using reanalaysis data and quantifying +uncertainty are provided as well. + +The general approach for estimating wake losses and quantifying uncertainty using bootstrapping +is based in part on the following publications: + +1. Barthelmie, R. J. and Jensen, L. E. Evaluation of wind farm efficiency and wind turbine wakes + at the Nysted offshore wind farm, *Wind Energy* 13(6):573–586 (2010). + https://doi.org/10.1002/we.408. +2. Nygaard, N. G. Systematic quantification of wake model uncertainty. Proc. EWEA Offshore, + Copenhagen, Denmark, March 10-12 (2015). +3. Walker, K., Adams, N., Gribben, B., Gellatly, B., Nygaard, N. G., Henderson, A., Marchante + Jimémez, M., Schmidt, S. R., Rodriguez Ruiz, J., Paredes, D., Harrington, G., Connell, N., + Peronne, O., Cordoba, M., Housley, P., Cussons, R., Håkansson, M., Knauer, A., and Maguire, + E.: An evaluation of the predictive accuracy of wake effects models for offshore wind farms. + *Wind Energy* 19(5):979–996 (2016). https://doi.org/10.1002/we.1871. + +The corrections for freestream wind speed heterogeneity are based in part on the approach +presented in: +4. Kassebaum, J. Wake Validation Through SCADA Data Analysis. Proc. American Clean Power Resource + & Project Energy Assessment Virtual Summit 2021 (2021). +""" # noqa: RUF002 from __future__ import annotations import random import itertools from copy import deepcopy -from typing import Callable +from collections.abc import Callable import attrs import numpy as np @@ -45,15 +46,11 @@ from sklearn.linear_model import LinearRegression from openoa.plant import PlantData, convert_to_list -from openoa.utils import plot, filters, power_curve -from openoa.utils import met_data_processing as met +from openoa.utils import plot, filters, power_curve, met_data_processing as met from openoa.schema import FromDictMixin, ResetValuesMixin from openoa.logging import logging, logged_method_call -from openoa.analysis._analysis_validators import ( - validate_UQ_input, - validate_half_closed_0_1_right, - validate_reanalysis_selections, -) +from openoa.analysis._analysis_validators import validate_UQ_input, validate_reanalysis_selections + logger = logging.getLogger(__name__) NDArrayFloat = npt.NDArray[np.float64] @@ -62,8 +59,7 @@ @define(auto_attribs=True) class WakeLosses(FromDictMixin, ResetValuesMixin): - """ - A serial implementation of a method for estimating wake losses from SCADA data. Wake losses are + """A serial implementation of a method for estimating wake losses from SCADA data. Wake losses are estimated for the entire wind plant as well as for each individual turbine for a) the period of record for which data are available, and b) the estimated long-term wind conditions the wind plant will experience based on historical reanalysis wind resource data. @@ -242,6 +238,7 @@ class WakeLosses(FromDictMixin, ResetValuesMixin): bin_count_thresh_lin_reg (int, optional): The minimum number of samples required in a wind speed bin to include when finding linear regression from SCADA freestream wind speeds to reanalysis wind speeds. Defaults to 50. + """ plant: PlantData = field(converter=deepcopy, validator=attrs.validators.instance_of(PlantData)) @@ -362,15 +359,13 @@ def check_reanalysis_products(self, attribute: attrs.Attribute, value: list[str] @logged_method_call def __attrs_post_init__(self): - """ - Initialize logging and post-initialization setup steps. - """ + """Initialize logging and post-initialization setup steps.""" logger.info("Initializing WakeLosses analysis object") - if self.wind_direction_data_type == "scada": + if self.wind_direction_data_type == "scada": # noqa: SIM102 if {"WakeLosses-scada", "all"}.intersection(self.plant.analysis_type) == set(): self.plant.analysis_type.append("WakeLosses-scada") - if self.wind_direction_data_type == "tower": + if self.wind_direction_data_type == "tower": # noqa: SIM102 if {"WakeLosses-tower", "all"}.intersection(self.plant.analysis_type) == set(): self.plant.analysis_type.append("WakeLosses-tower") @@ -426,22 +421,22 @@ def run( freestream_sector_width: float | None = None, freestream_power_method: str | None = None, freestream_wind_speed_method: str | None = None, - correct_for_derating: bool | None = None, derating_filter_wind_speed_start: float | None = None, max_power_filter: float | None = None, wind_bin_mad_thresh: float | None = None, - correct_for_ws_heterogeneity: bool | None = None, ws_speedup_factor_map: pd.DataFrame | str | None = None, wd_bin_width_LT_corr: float | None = None, ws_bin_width_LT_corr: float | None = None, num_years_LT: int | None = None, - assume_no_wakes_high_ws_LT_corr: bool | None = None, no_wakes_ws_thresh_LT_corr: float | None = None, min_ws_bin_lin_reg: float | None = None, bin_count_thresh_lin_reg: int | None = None, + *, + correct_for_derating: bool | None = None, + correct_for_ws_heterogeneity: bool | None = None, + assume_no_wakes_high_ws_LT_corr: bool | None = None, ): - """ - Estimates wake losses by comparing wind plant energy production to energy production of the + """Estimates wake losses by comparing wind plant energy production to energy production of the turbines identified as operating in freestream conditions. Wake losses are expressed as a fractional loss (e.g., 0.05 indicates a wake loss values of 5%). @@ -451,6 +446,9 @@ def run( Args: num_sim (int, optional): Number of Monte Carlo iterations to perform. Only used if :py:attr:`UQ` = True. Defaults to 100. + reanalysis_products (:obj:`list`, optional): List of reanalysis products to use for long-term + correction. If UQ = True, a single product will be selected form this list each Monte + Carlo iteration. Defaults to ["merra2", "era5"]. wd_bin_width (float, optional): Wind direction bin size when identifying freestream wind turbines (degrees). Defaults to 5 degrees. freestream_sector_width (tuple | float, optional): Wind direction sector size to use when @@ -538,6 +536,7 @@ def run( bin_count_thresh_lin_reg (int, optional): The minimum number of samples required in a wind speed bin to include when finding linear regression from SCADA freestream wind speeds to reanalysis wind speeds. Defaults to 50. + """ initial_parameters = {} # Assign default parameter values depending on whether UQ is performed @@ -1097,11 +1096,9 @@ def run( @logged_method_call def _setup_monte_carlo_inputs(self): - """ - Create and populate the data frame defining the Monte Carlo simulation parameters. This + """Create and populate the data frame defining the Monte Carlo simulation parameters. This data frame is stored as ``self.inputs``. """ - if self.UQ: inputs = { "reanalysis_product": random.choices(self.reanalysis_products, k=self.num_sim), @@ -1187,12 +1184,10 @@ def _setup_monte_carlo_inputs(self): @logged_method_call def _calculate_aggregate_dataframe(self): - """ - Creates a data frame with relevant scada columns, plant-level columns, and reanalysis + """Creates a data frame with relevant scada columns, plant-level columns, and reanalysis variables to be used for the wake loss analysis. The reference mean wind direction is then added to the data frame. """ - # keep relevant SCADA columns, create a unique time index and two-level turbine variable columns # (variable name and turbine asset_id) @@ -1225,12 +1220,10 @@ def _calculate_aggregate_dataframe(self): @logged_method_call def _calculate_mean_wind_direction(self): - """ - Calculates the mean wind direction at each time step using the specified wind direction column for the + """Calculates the mean wind direction at each time step using the specified wind direction column for the specified subset of turbines or met towers. This reference mean wind direction is added to the plant-level data frame. """ - if self.wind_direction_data_type == "scada": self.aggregate_df["wind_direction_ref"] = met.circular_mean( self.aggregate_df[self.wind_direction_col][self.wind_direction_asset_ids], axis=1 @@ -1244,10 +1237,7 @@ def _calculate_mean_wind_direction(self): @logged_method_call def _include_reanal_data(self): - """ - Combines reanalysis data columns with the aggregate data frame for use in long-term correction. - """ - + """Combines reanalysis data columns with the aggregate data frame for use in long-term correction.""" # combine all wind speed and wind direction reanalysis variables into aggregate data frame for product in self.reanalysis_products: @@ -1261,16 +1251,14 @@ def _include_reanal_data(self): df_rean = df_rean.add_suffix(f"_{product}") df_rean = df_rean[df_rean.index.isin(self.aggregate_df.index)] - self.aggregate_df[[col for col in df_rean.columns]] = df_rean + self.aggregate_df[df_rean.columns] = df_rean @logged_method_call def _get_speedup_factors(self): - """ - Loads table of wind speed speedup factors as a function of wind direction for each + """Loads table of wind speed speedup factors as a function of wind direction for each turbine, then creates a speedup factor column for each turbine by linearly interpolating the speedup factors using the reference wind direction. """ - if type(self.ws_speedup_factor_map) is str: df_ws_speedup_factor_map = pd.read_csv(self.ws_speedup_factor_map) else: @@ -1286,15 +1274,15 @@ def _get_speedup_factors(self): # 0 to 360 degrees if wd_last < 360: df_ws_speedup_factor_map = pd.concat([df_ws_speedup_factor_map, first_row], axis=0) - df_ws_speedup_factor_map.iloc[ - -1, df_ws_speedup_factor_map.columns.get_loc("wd") - ] += 360.0 + df_ws_speedup_factor_map.iloc[-1, df_ws_speedup_factor_map.columns.get_loc("wd")] += ( + 360.0 + ) if wd_first > 0: df_ws_speedup_factor_map = pd.concat([last_row, df_ws_speedup_factor_map], axis=0) - df_ws_speedup_factor_map.iloc[ - 0, df_ws_speedup_factor_map.columns.get_loc("wd") - ] -= 360.0 + df_ws_speedup_factor_map.iloc[0, df_ws_speedup_factor_map.columns.get_loc("wd")] -= ( + 360.0 + ) df_ws_speedup_factor_map = df_ws_speedup_factor_map.reset_index(drop=True) @@ -1307,11 +1295,9 @@ def _get_speedup_factors(self): @logged_method_call def _identify_derating(self): - """ - Estimates whether each turbine is derated, curtailed, or otherwise not operating for each time stamp based on + """Estimates whether each turbine is derated, curtailed, or otherwise not operating for each time stamp based on power curve filtering. A derated flag is then added to the aggregate data frame for each turbine. """ - for t in self.turbine_ids: # Apply window range filter to flag samples for which wind speed is greater than a threshold and power is # below 1% of rated power @@ -1376,8 +1362,7 @@ def _identify_derating(self): @logged_method_call def _apply_LT_correction(self): - """ - Estimates long term-corrected wake losses by binning wake losses by wind direction and wind + """Estimates long term-corrected wake losses by binning wake losses by wind direction and wind speed and weighting by bin frequencies from long-term historical reanalysis data. Returns: @@ -1386,6 +1371,7 @@ def _apply_LT_correction(self): term-corrected wake losses, and arrays containing the long-term corrected plant and turbine-level wake losses as well as the normalized wind plant energy production binned by wind direction + """ # First, create hourly data frame for LT correction to match resolution of reanalysis data df_1hr = self.aggregate_df_sample[ @@ -1506,7 +1492,7 @@ def _apply_LT_correction(self): ) df_1hr_bin.loc[ fill_inds, [("actual_plant_power", ""), ("potential_plant_power", "")] - ] = (self.plant.metadata.capacity * 1e3) + ] = self.plant.metadata.capacity * 1e3 df_1hr_bin.loc[ fill_inds, [("WTUR_W", t) for t in self.turbine_ids] @@ -1601,19 +1587,19 @@ def _apply_LT_correction(self): def plot_wake_losses_by_wind_direction( self, - plot_norm_energy: bool = True, - turbine_id: str = None, + turbine_id: str | None = None, xlim: tuple[float, float] = (None, None), ylim_efficiency: tuple[float, float] = (None, None), ylim_energy: tuple[float, float] = (None, None), - return_fig: bool = False, figure_kwargs: dict | None = None, plot_kwargs_line: dict | None = None, plot_kwargs_fill: dict | None = None, legend_kwargs: dict | None = None, + *, + plot_norm_energy: bool = True, + return_fig: bool = False, ): - """ - Plots wake losses in the form of wind farm efficiency as well as normalized wind plant energy + """Plots wake losses in the form of wind farm efficiency as well as normalized wind plant energy production for both the period of record and with the long-term correction as a function of wind direction. @@ -1643,13 +1629,14 @@ def plot_wake_losses_by_wind_direction( legend_kwargs (:obj:`dict`, optional): Additional legend keyword arguments that are passed to ``ax.legend()`` for the wind farm efficiency and, if `plot_norm_energy` is True, energy distributions subplots. Defaults to None. + Returns: None | tuple[matplotlib.pyplot.Figure, matplotlib.pyplot.Axes] | tuple[matplotlib.pyplot.Figure, tuple [matplotlib.pyplot.Axes, matplotlib.pyplot.Axes]]: If :py:attr:`return_fig` is True, then the figure and axes object(s), corresponding to the wake loss plot or, if :py:attr:`plot_norm_energy` is True, wake loss and normalized energy plots, are returned for further tinkering/saving. - """ + """ if figure_kwargs is None: figure_kwargs = {} if plot_kwargs_line is None: @@ -1702,19 +1689,19 @@ def plot_wake_losses_by_wind_direction( def plot_wake_losses_by_wind_speed( self, - plot_norm_energy: bool = True, - turbine_id: str = None, + turbine_id: str | None = None, xlim: tuple[float, float] = (None, None), ylim_efficiency: tuple[float, float] = (None, None), ylim_energy: tuple[float, float] = (None, None), - return_fig: bool = False, figure_kwargs: dict | None = None, plot_kwargs_line: dict | None = None, plot_kwargs_fill: dict | None = None, legend_kwargs: dict | None = None, + *, + plot_norm_energy: bool = True, + return_fig: bool = False, ): - """ - Plots wake losses in the form of wind farm efficiency as well as normalized wind plant energy + """Plots wake losses in the form of wind farm efficiency as well as normalized wind plant energy production for both the period of record and with the long-term correction as a function of wind speed. @@ -1744,13 +1731,14 @@ def plot_wake_losses_by_wind_speed( legend_kwargs (:obj:`dict`, optional): Additional legend keyword arguments that are passed to ``ax.legend()`` for the wind farm efficiency and, if :py:attr:`plot_norm_energy` is True, energy distributions subplots. Defaults to None. + Returns: None | tuple[matplotlib.pyplot.Figure, matplotlib.pyplot.Axes] | tuple[matplotlib.pyplot.Figure, tuple [matplotlib.pyplot.Axes, matplotlib.pyplot.Axes]]: If :py:attr:`return_fig` is True, then the figure and axes object(s), corresponding to the wake loss plot or, if :py:attr:`plot_norm_energy` is True, wake loss and normalized energy plots, are returned for further tinkering/saving. - """ + """ if figure_kwargs is None: figure_kwargs = {} if plot_kwargs_line is None: @@ -1853,12 +1841,11 @@ def plot_wake_losses_by_wind_speed( ) -def create_WakeLosses( +def create_WakeLosses( # ruff: ignore[D103] project: PlantData, wind_direction_col: str = __defaults_wind_direction_col, wind_direction_data_type: str = __defaults_wind_direction_data_type, wind_direction_asset_ids: list[str] = __defaults_wind_direction_asset_ids, - UQ: bool = __defaults_UQ, num_sim: int = __defaults_num_sim, start_date: str | pd.Timestamp = __defaults_start_date, end_date: str | pd.Timestamp = __defaults_end_date, @@ -1868,15 +1855,17 @@ def create_WakeLosses( freestream_sector_width: float = __defaults_freestream_sector_width, freestream_power_method: str = __defaults_freestream_power_method, freestream_wind_speed_method: str = __defaults_freestream_wind_speed_method, - correct_for_derating: bool = __defaults_correct_for_derating, - derating_filter_wind_speed_start: float = __defaults_derating_filter_wind_speed_start, max_power_filter: float = __defaults_max_power_filter, wind_bin_mad_thresh: float = __defaults_wind_bin_mad_thresh, wd_bin_width_LT_corr: float = __defaults_wd_bin_width_LT_corr, ws_bin_width_LT_corr: float = __defaults_ws_bin_width_LT_corr, num_years_LT: int = __defaults_num_years_LT, - assume_no_wakes_high_ws_LT_corr: bool = __defaults_assume_no_wakes_high_ws_LT_corr, no_wakes_ws_thresh_LT_corr: float = __defaults_no_wakes_ws_thresh_LT_corr, + *, + UQ: bool = __defaults_UQ, + correct_for_derating: bool = __defaults_correct_for_derating, + derating_filter_wind_speed_start: float = __defaults_derating_filter_wind_speed_start, + assume_no_wakes_high_ws_LT_corr: bool = __defaults_assume_no_wakes_high_ws_LT_corr, ) -> WakeLosses: return WakeLosses( plant=project, diff --git a/openoa/analysis/yaw_misalignment.py b/openoa/analysis/yaw_misalignment.py index fc1b44e76..58ae88bc3 100644 --- a/openoa/analysis/yaw_misalignment.py +++ b/openoa/analysis/yaw_misalignment.py @@ -1,36 +1,37 @@ -# This class contains analytical routines for estimating static yaw misalignment for a list of -# specified wind turbines. For a series of wind speed bins, the yaw misalignment is estimated for -# each turbine by identifying the difference between the wind vane angle where power performance -# (either power or normalized coefficient of power) is maximized and the mean wind vane angle for -# the turbine. The wind vane angle where power is maximized is found by fitting a cosine exponent -# curve to the binned power performance values as a function of wind vane angle. The offset of the -# best-fit cosine curve is treated as the wind vane angle where power is maximized. In addition to -# static yaw misalignment estimates for each wind speed bin, the mean yaw misalignment value -# averaged over all wind speed bins is calculated. -# -# Many parts of this method are based on or inspired by the following publications: - -# 1. Bao, Y., Yang, Q., Fu, L., Chen, Q., Cheng, C., and Sun, Y. Identification of Yaw Error -# Inherent Misalignment for Wind Turbine Based on SCADA Data: A Data Mining Approach. Proc. 12th -# Asian Control Conference (ASCC), Kitakyushu, Japan, June 9-12 (2019). 1095-1100. -# 2. Xue, J. and Wang, L. Online data-driven approach of yaw error estimation and correction of -# horizontal axis wind turbine. *IET J. Eng.* 2019(18):4937–4940 (2019). -# https://doi.org/10.1049/joe.2018.9293. -# 3. Astolfi, D., Castellani, F., and Terzi, L. An Operation Data-Based Method for the Diagnosis of -# Zero-Point Shift of Wind Turbines Yaw Angle. *J. Solar Energy Engineering* 142(2):024501 -# (2020). https://doi.org/10.1115/1.4045081. -# 4. Jing, B., Qian, Z., Pei, Y., Zhang, L., and Yang, T. Improving wind turbine efficiency through -# detection and calibration of yaw misalignment. *Renewable Energy* 160:1217-1227 (2020). -# https://doi.org/10.1016/j.renene.2020.07.063. -# 5. Gao, L. and Hong, J. Data-driven yaw misalignment correction for utility-scale wind turbines. -# *J. Renewable Sustainable Energy* 13(6):063302 (2021). https://doi.org/10.1063/5.0056671. - -# WARNING: This is a relatively simple method that has not yet been validated using data from wind -# turbines with known static yaw misalignments. Therefore, the results should be treated with -# caution. One known issue is that the method currently relies on nacelle wind speed measurements -# to determine the power performance as a function of wind vane angle. If the measured wind speed -# is affected by the amount of yaw misalignment, potential biases can exist in the estimated static -# yaw misalignment values. +"""This class contains analytical routines for estimating static yaw misalignment for a list of +specified wind turbines. For a series of wind speed bins, the yaw misalignment is estimated for +each turbine by identifying the difference between the wind vane angle where power performance +(either power or normalized coefficient of power) is maximized and the mean wind vane angle for +the turbine. The wind vane angle where power is maximized is found by fitting a cosine exponent +curve to the binned power performance values as a function of wind vane angle. The offset of the +best-fit cosine curve is treated as the wind vane angle where power is maximized. In addition to +static yaw misalignment estimates for each wind speed bin, the mean yaw misalignment value +averaged over all wind speed bins is calculated. + +Many parts of this method are based on or inspired by the following publications: + +1. Bao, Y., Yang, Q., Fu, L., Chen, Q., Cheng, C., and Sun, Y. Identification of Yaw Error + Inherent Misalignment for Wind Turbine Based on SCADA Data: A Data Mining Approach. Proc. 12th + Asian Control Conference (ASCC), Kitakyushu, Japan, June 9-12 (2019). 1095-1100. +2. Xue, J. and Wang, L. Online data-driven approach of yaw error estimation and correction of + horizontal axis wind turbine. *IET J. Eng.* 2019(18):4937–4940 (2019). + https://doi.org/10.1049/joe.2018.9293. +3. Astolfi, D., Castellani, F., and Terzi, L. An Operation Data-Based Method for the Diagnosis of + Zero-Point Shift of Wind Turbines Yaw Angle. *J. Solar Energy Engineering* 142(2):024501 + (2020). https://doi.org/10.1115/1.4045081. +4. Jing, B., Qian, Z., Pei, Y., Zhang, L., and Yang, T. Improving wind turbine efficiency through + detection and calibration of yaw misalignment. *Renewable Energy* 160:1217-1227 (2020). + https://doi.org/10.1016/j.renene.2020.07.063. +5. Gao, L. and Hong, J. Data-driven yaw misalignment correction for utility-scale wind turbines. + *J. Renewable Sustainable Energy* 13(6):063302 (2021). https://doi.org/10.1063/5.0056671. + +WARNING: This is a relatively simple method that has not yet been validated using data from wind +turbines with known static yaw misalignments. Therefore, the results should be treated with +caution. One known issue is that the method currently relies on nacelle wind speed measurements +to determine the power performance as a function of wind vane angle. If the measured wind speed +is affected by the amount of yaw misalignment, potential biases can exist in the estimated static +yaw misalignment values. +""" # noqa: RUF002 from __future__ import annotations @@ -50,6 +51,7 @@ from openoa.logging import logging, logged_method_call from openoa.analysis._analysis_validators import validate_UQ_input, validate_half_closed_0_1_right + logger = logging.getLogger(__name__) NDArrayFloat = npt.NDArray[np.float64] plot.set_styling() @@ -64,16 +66,17 @@ def cos_curve(x, A, Offset, cos_exp): Offset (:obj:`float`): The yaw misaligment offset at which the cosine exponent curve is maximized in degrees. cos_exp (:obj:`float`): The exponent to which the cosine curve is raised. + Returns: :obj:`float`: The value of the cosine exponent curve for the provided yaw misalignment. + """ return A * np.cos((np.pi / 180) * (x - Offset)) ** cos_exp @define(auto_attribs=True) class StaticYawMisalignment(FromDictMixin, ResetValuesMixin): - """ - A method for estimating static yaw misalignment for different wind speed bins for each specified + """A method for estimating static yaw misalignment for different wind speed bins for each specified wind turbine as well as the average static yaw misalignment over all wind speed bins using turbine-level SCADA data. @@ -155,6 +158,7 @@ class StaticYawMisalignment(FromDictMixin, ResetValuesMixin): use_power_coeff (bool, optional): If True, power performance as a function of wind vane angle will be quantified by normalizing power by the cube of the wind speed, approximating the power coefficient. If False, only power will be used. Defaults to False. + """ plant: PlantData = field(converter=deepcopy, validator=attrs.validators.instance_of(PlantData)) @@ -223,9 +227,7 @@ class StaticYawMisalignment(FromDictMixin, ResetValuesMixin): @logged_method_call def __attrs_post_init__(self): - """ - Initialize logging and post-initialization setup steps. - """ + """Initialize logging and post-initialization setup steps.""" if {"StaticYawMisalignment", "all"}.intersection(self.plant.analysis_type) == set(): self.plant.analysis_type.append("StaticYawMisalignment") @@ -257,10 +259,10 @@ def run( min_power_filter: float | None = None, max_power_filter: float | None = None, power_bin_mad_thresh: float | None = None, + *, use_power_coeff: bool | None = None, ): - """ - Estimates static yaw misalignment for each wind speed bin for each specified wind turbine. + """Estimates static yaw misalignment for each wind speed bin for each specified wind turbine. After performing power curve filtering to remove timestamps when pitch angle is above a threshold or the turbine is operating abnormally, best-fit cosine curves are found for binned power performance vs. wind vane angle for each wind speed bin and turbine. The @@ -305,6 +307,7 @@ def run( use_power_coeff (bool, optional): If True, power performance as a function of wind vane angle will be quantified by normalizing power by the cube of the wind speed, approximating the power coefficient. If False, only power will be used. Defaults to False. + """ initial_parameters = {} if num_sim is not None: @@ -422,12 +425,10 @@ def run( @logged_method_call def _setup_monte_carlo_inputs(self): - """ - Create and populate the data frame defining the Monte Carlo simulation parameters. This + """Create and populate the data frame defining the Monte Carlo simulation parameters. This data frame is stored as self.inputs. Variables used to save intermediate variables and final results are also initiated. """ - if self.UQ: inputs = { "power_bin_mad_thresh": np.random.randint( @@ -494,8 +495,7 @@ def _setup_monte_carlo_inputs(self): @logged_method_call def _remove_power_curve_outliers(self, turbine_id): - """ - Removes power curve outliers for a specific turbine by removing timestamps where the pitch + """Removes power curve outliers for a specific turbine by removing timestamps where the pitch angle is above a threshold and timestamps where the wind speed is more than a specific threshold from the median wind speed in each power bin. The filtered turbine data frame is meant to include timestamps when the turbine is operating normally in below-rated @@ -503,8 +503,8 @@ def _remove_power_curve_outliers(self, turbine_id): Args: turbine_id (str): The name of the turbine for which power curve outlier removal will be performed. - """ + """ # Limit to pitch angles below the specified threshold self._df_turb = self._df_turb.loc[self._df_turb["WROT_BlPthAngVal"] <= self.pitch_thresh] @@ -528,8 +528,7 @@ def _remove_power_curve_outliers(self, turbine_id): @logged_method_call def _estimate_static_yaw_misalignment(self): - """ - Estimates static yaw misalignment for a single turbine and wind speed bin by fitting a + """Estimates static yaw misalignment for a single turbine and wind speed bin by fitting a cosine curve to the binned power performance vs. wind vane angle. Yaw misalignment is estimated as the difference between the wind vane angle where power is maximized based on the best-fit cosine curve and the mean wind vane angle. @@ -539,8 +538,8 @@ def _estimate_static_yaw_misalignment(self): mean wind vane angle, and arrays containing the best-fit cosine curve parameters (magnitude, offset (degrees), and cosine exponent) and power performance values binned by wind vane angle. - """ + """ self._df_turb_ws["vane_bin"] = self.vane_bin_width * np.round( self._df_turb_ws["WMET_HorWdDirRel"].values / self.vane_bin_width ) @@ -583,15 +582,16 @@ def _estimate_static_yaw_misalignment(self): def plot_yaw_misalignment_by_turbine( self, - turbine_ids: list[str] = None, + turbine_ids: list[str] | None = None, xlim: tuple[float, float] = (None, None), ylim: tuple[float, float] = (None, None), - return_fig: bool = False, figure_kwargs: dict | None = None, plot_kwargs_curve: dict | None = None, plot_kwargs_line: dict | None = None, plot_kwargs_fill: dict | None = None, legend_kwargs: dict | None = None, + *, + return_fig: bool = False, ): """Plots power performance vs. wind vane angle along with the best-fit cosine curve for each wind speed bin for each turbine specified. The mean wind vane angle and the wind vane @@ -622,14 +622,15 @@ def plot_yaw_misalignment_by_turbine( intervals for power performance vs. wind vane. Defaults to None. legend_kwargs (:obj:`dict`, optional): Additional legend keyword arguments that are passed to ``ax.legend()`` for the power performance vs. wind vane plots. Defaults to None. + Returns: None | dict of tuple[matplotlib.pyplot.Figure, matplotlib.pyplot.Axes]: If :py:attr:`return_fig` is True, then a dictionary containing the figure and axes object(s) corresponding to the yaw misalignment plots for each turbine are returned for further tinkering/saving. The turbine names in the `turbine_ids` aregument are the dicitonary keys. - """ + """ if self.use_power_coeff: power_performance_label = "Normalized Cp (-)" else: @@ -703,10 +704,9 @@ def plot_yaw_misalignment_by_turbine( __defaults_use_power_coeff = StaticYawMisalignment.__attrs_attrs__.use_power_coeff.default -def create_StaticYawMisalignment( +def create_StaticYawMisalignment( # ruff: ignore[D103] project: PlantData, turbine_ids: list[str] = __defaults_turbine_ids, - UQ: bool = __defaults_UQ, num_sim: int = __defaults_num_sim, ws_bins: list[float] = __defaults_ws_bins, ws_bin_width: float = __defaults_ws_bin_width, @@ -718,6 +718,8 @@ def create_StaticYawMisalignment( min_power_filter: float = __defaults_min_power_filter, max_power_filter: float | tuple[float, float] = __defaults_max_power_filter, power_bin_mad_thresh: float | tuple[float, float] = __defaults_power_bin_mad_thresh, + *, + UQ: bool = __defaults_UQ, use_power_coeff: bool = __defaults_use_power_coeff, ) -> StaticYawMisalignment: return StaticYawMisalignment( diff --git a/openoa/logging.py b/openoa/logging.py index 003009c6e..b79eb9c44 100644 --- a/openoa/logging.py +++ b/openoa/logging.py @@ -1,3 +1,5 @@ +"""Provides standardized logging.""" + import os import json import logging @@ -7,12 +9,13 @@ def setup_logging( - console: bool = True, level: str = "WARNING", configuration: str = "logging.json", env_key="LOG_CFG", + *, + console: bool = True, ): - """Setup logging configuration""" + """Setup logging configuration.""" if (value := os.getenv(env_key, None)) is not None: configuration = value configuration = Path(configuration).resolve() @@ -22,10 +25,12 @@ def setup_logging( else: logging.basicConfig(level=level) - logging.captureWarnings(True) + logging.captureWarnings(capture=True) def logged_method_call(the_method, msg="call"): + """Logs a method call.""" + @wraps(the_method) def _wrapper(self, *args, **kwargs): logger = logging.getLogger(the_method.__module__) @@ -37,6 +42,8 @@ def _wrapper(self, *args, **kwargs): def logged_function_call(the_function, msg="call"): + """Logs a function.""" + @wraps(the_function) def _wrapper(*args, **kwargs): logger = logging.getLogger(the_function.__module__) diff --git a/openoa/plant.py b/openoa/plant.py index fa1bc10cd..dd0c93115 100644 --- a/openoa/plant.py +++ b/openoa/plant.py @@ -1,10 +1,12 @@ +"""Provides the :py:class:`PlantData` object and primary setup routines.""" + from __future__ import annotations import sys import logging import itertools -from typing import Callable, Optional, Sequence from pathlib import Path +from collections.abc import Callable, Sequence import yaml import attrs @@ -23,6 +25,7 @@ from openoa.utils.metadata_fetch import attach_eia_data from openoa.utils.unit_conversion import convert_power_to_energy + setup_logging(level="WARNING") logger = logging.getLogger(__name__) @@ -34,7 +37,7 @@ @logged_method_call def _analysis_filter( - error_dict: dict, metadata: PlantMetaData, analysis_types: list[str] = ["all"] + error_dict: dict, metadata: PlantMetaData, analysis_types: list[str] | None = None ) -> dict: """Filters the errors found by the analysis requirements provided by the :py:attr:`analysis_types`. @@ -53,7 +56,10 @@ def _analysis_filter( Returns: dict: The missing column, bad dtype, and incorrect timestamp frequency errors corresponding to the user's analysis types. + """ + if analysis_types is None: + analysis_types = ["all"] if "all" in analysis_types: return error_dict @@ -101,7 +107,7 @@ def _analysis_filter( @logged_method_call def _compose_error_message( - error_dict: dict, metadata: PlantMetaData, analysis_types: list[str] = ["all"] + error_dict: dict, metadata: PlantMetaData, analysis_types: list[str] | None = None ) -> str: """Takes a dictionary of error messages from the ``PlantData`` validation routines, filters out errors unrelated to the intended analysis types, and creates a @@ -116,7 +122,10 @@ def _compose_error_message( Returns: str: The human-readable error message breakdown. + """ + if analysis_types is None: + analysis_types = ["all"] if analysis_types == [None]: return "" @@ -148,6 +157,7 @@ def _compose_error_message( def frequency_validator( actual_freq: str | int | float | None, desired_freq: str | set[str] | None, + *, exact: bool, ) -> bool: """Helper function to check if the actual datetime stamp frequency is valid compared @@ -164,6 +174,7 @@ def frequency_validator( Returns: (:obj:`bool`): If the actual datetime frequency is sufficient, per the match requirements. + """ if desired_freq is None: return True @@ -190,7 +201,7 @@ def frequency_validator( return actual_freq < max(desired_freq) -def convert_to_list( +def convert_to_list( # ruff: ignore[D417] <- "manipulation" is registering as "ma\nipulation" value: Sequence | str | int | float | None, manipulation: Callable | None = None, ) -> list: @@ -198,14 +209,15 @@ def convert_to_list( to a list of elements. Args: - value(:obj:`Sequence` | :obj:`str` | :obj:`int` | :obj:`float`): The unknown element to be + value (:obj:`Sequence` | :obj:`str` | :obj:`int` | :obj:`float`): The unknown element to be converted to a list of element(s). - manipulation(:obj:`Callable` | :obj:`None`) A function to be performed upon the individual elements, by default None. + manipulation (:obj:`Callable` | :obj:`None`) A function to be performed upon the individual + elements, by default None. Returns: - (:ojb:`list`): The new list of elements. - """ + (:obj:`list`): The new list of elements. + """ if isinstance(value, (str, int, float)) or value is None: value = [value] if manipulation is not None: @@ -214,7 +226,7 @@ def convert_to_list( @logged_method_call -def column_validator(df: pd.DataFrame, column_names={}) -> None | list[str]: +def column_validator(df: pd.DataFrame, column_names: dict | None = None) -> list[str] | None: """Validates that the column names exist as provided for each expected column. Args: @@ -225,7 +237,10 @@ def column_validator(df: pd.DataFrame, column_names={}) -> None | list[str]: Returns: None | list[str]: A list of error messages that can be raised at a later step in the validation process. + """ + if column_names is None: + column_names = {} try: missing = set(column_names.values()).difference(df.columns) except AttributeError: @@ -237,7 +252,7 @@ def column_validator(df: pd.DataFrame, column_names={}) -> None | list[str]: @logged_method_call -def dtype_converter(df: pd.DataFrame, column_types={}) -> list[str]: +def dtype_converter(df: pd.DataFrame, column_types: dict | None = None) -> list[str]: """Converts the columns provided in :py:attr:`column_types` of :py:attr:`df` to the appropriate data type. @@ -249,18 +264,21 @@ def dtype_converter(df: pd.DataFrame, column_types={}) -> list[str]: Returns: None | list[str]: List of error messages that were encountered in the conversion process that will be raised at another step of the data validation. + """ + if column_types is None: + column_types = {} errors = [] for column, new_type in column_types.items(): if new_type in (np.datetime64, pd.DatetimeIndex): try: df[column] = pd.DatetimeIndex(df[column]) - except Exception as e: # noqa: disable=E722 + except Exception: errors.append(column) continue try: df[column] = df[column].astype(new_type) - except: # noqa: disable=E722 + except: # ruff: ignore[E722] errors.append(column) return errors @@ -278,6 +296,7 @@ def load_to_pandas(data: str | Path | pd.DataFrame) -> pd.DataFrame | None: Returns: pd.DataFrame | None: The passed ``None`` or the converted pandas DataFrame object. + """ if data is None: return data @@ -302,6 +321,7 @@ def load_to_pandas_dict( Returns: dict[str, pd.DataFrame] | None: The passed ``None`` or the converted ``pd.DataFrame`` object. + """ if data is None: return data @@ -311,19 +331,20 @@ def load_to_pandas_dict( @logged_method_call -def rename_columns(df: pd.DataFrame, col_map: dict, reverse: bool = True) -> pd.DataFrame: +def rename_columns(df: pd.DataFrame, col_map: dict, *, reverse: bool = True) -> pd.DataFrame: """Renames the pandas DataFrame columns using col_map. Intended to be used in conjunction with the a data objects meta data column mapping (``reverse=True``). - Args: + Args: df (pd.DataFrame): The DataFrame to have its columns remapped. col_map (dict): Dictionary of existing column names and new column names. reverse (bool, optional): True, if the new column names are the keys (using the xxMetaData.col_map as input), or False, if the current column names are the values (original column names). Defaults to True. - Returns: + Returns: pd.DataFrame: Input DataFrame with remapped column names. + """ if reverse: col_map = {v: k for k, v in col_map.items()} @@ -406,6 +427,7 @@ class PlantData: Raises: ValueError: Raised if any analysis specific validation checks don't pass with an error message highlighting the appropriate issues. + """ log_level: str = field(default="WARNING", converter=set_log_level) @@ -417,22 +439,20 @@ class PlantData: ) analysis_type: list[str] | None = field( default=None, - converter=convert_to_list, # noqa: F821 + converter=convert_to_list, validator=attrs.validators.deep_iterable( iterable_validator=attrs.validators.instance_of(list), - member_validator=attrs.validators.in_([*ANALYSIS_REQUIREMENTS] + ["all", None]), + member_validator=attrs.validators.in_([*ANALYSIS_REQUIREMENTS, "all", None]), ), on_setattr=[attrs.setters.convert, attrs.setters.validate], ) - scada: pd.DataFrame | None = field(default=None, converter=load_to_pandas) # noqa: F821 - meter: pd.DataFrame | None = field(default=None, converter=load_to_pandas) # noqa: F821 - tower: pd.DataFrame | None = field(default=None, converter=load_to_pandas) # noqa: F821 - status: pd.DataFrame | None = field(default=None, converter=load_to_pandas) # noqa: F821 - curtail: pd.DataFrame | None = field(default=None, converter=load_to_pandas) # noqa: F821 - asset: pd.DataFrame | None = field(default=None, converter=load_to_pandas) # noqa: F821 - reanalysis: dict[str, pd.DataFrame] | None = field( - default=None, converter=load_to_pandas_dict # noqa: F821 - ) + scada: pd.DataFrame | None = field(default=None, converter=load_to_pandas) + meter: pd.DataFrame | None = field(default=None, converter=load_to_pandas) + tower: pd.DataFrame | None = field(default=None, converter=load_to_pandas) + status: pd.DataFrame | None = field(default=None, converter=load_to_pandas) + curtail: pd.DataFrame | None = field(default=None, converter=load_to_pandas) + asset: pd.DataFrame | None = field(default=None, converter=load_to_pandas) + reanalysis: dict[str, pd.DataFrame] | None = field(default=None, converter=load_to_pandas_dict) # No user initialization required for attributes defined below here # Error catching in validation @@ -490,6 +510,7 @@ def data_validator(self, instance: attrs.Attribute, value: pd.DataFrame | None) instance (:obj:`attrs.Attribute`): The ``attrs.Attribute`` details value (:obj:`pd.DataFrame | None`): The attribute's user-provided value. A dictionary of dataframes is expected for reanalysis data only. + """ name = instance.name if self.analysis_type == [None]: @@ -519,6 +540,7 @@ def reanalysis_validator( instance (:obj:`attrs.Attribute`): The :py:attr:`attrs.Attribute` details. value (:obj:`dict[str, pd.DataFrame]` | None): The attribute's user-provided value. A dictionary of dataframes is expected for reanalysis data only. + """ name = instance.name if value is not None: @@ -704,7 +726,7 @@ def _set_index_columns(self) -> None: def _unset_index_columns(self) -> None: """Resets the index for each of the data types. This is intended solely for the use with the :py:meth:`validate` to ensure the validation methods are able to find the index columns - in the column space + in the column space. """ if self.scada is not None: self.scada.reset_index(drop=False, inplace=True) @@ -728,23 +750,23 @@ def data_dict(self) -> dict[str, pd.DataFrame]: Returns: (:obj:`dict[str, pd.DataFrame]`): A mapping of the data type's name and the ``DataFrame``. + """ - values = dict( - scada=self.scada, - meter=self.meter, - tower=self.tower, - asset=self.asset, - status=self.status, - curtail=self.curtail, - reanalysis=self.reanalysis, - ) + values = { + "scada": self.scada, + "meter": self.meter, + "tower": self.tower, + "asset": self.asset, + "status": self.status, + "curtail": self.curtail, + "reanalysis": self.reanalysis, + } return values @logged_method_call def to_csv( self, save_path: str | Path, - with_openoa_col_names: bool = True, metadata: str = "metadata", scada: str = "scada", meter: str = "meter", @@ -753,6 +775,8 @@ def to_csv( status: str = "status", curtail: str = "curtail", reanalysis: str = "reanalysis", + *, + with_openoa_col_names: bool = True, ) -> None: """Saves all of the dataframe objects to a CSV file in the provided `save_path` directory. @@ -777,6 +801,7 @@ def to_csv( reanalysis (str, optional): Base file name (without extension) to be used for the reanalysis data, where each dataset will use the name provided to form the following file name: {save_path}/{reanalysis}_{name}. Defaults to "reanalysis". + """ save_path = Path(save_path).resolve() if not save_path.exists(): @@ -800,7 +825,7 @@ def to_csv( col_map["frequency"] = meta_obj.frequency meta[name] = col_map - with open((save_path / metadata).with_suffix(".yml"), "w") as f: + with (save_path / metadata).with_suffix(".yml").open("w") as f: yaml.safe_dump(meta, f, default_flow_style=False, sort_keys=False) if self.scada is not None: @@ -849,6 +874,7 @@ def _validate_column_names(self, category: str = "all") -> dict[str, list[str]]: Returns: dict[str, list[str]]: _description_ + """ column_map = self.metadata.column_map @@ -860,12 +886,12 @@ def _validate_column_names(self, category: str = "all") -> dict[str, list[str]]: if name == "reanalysis": # If no reanalysis data, get the default key from ReanalysisMetaData if df is None: - sub_name = [*column_map[name]][0] + sub_name = next(iter(column_map[name])) missing_cols[f"{name}-{sub_name}"] = column_validator( df, column_names=column_map[name][sub_name] ) continue - for sub_name, df in df.items(): + for sub_name, df in df.items(): # noqa: B020 logger.info(f"Validating column names in the {sub_name} {name} data") missing_cols[f"{name}-{sub_name}"] = column_validator( df, column_names=column_map[name][sub_name] @@ -887,6 +913,7 @@ def _validate_dtypes(self, category: str = "all") -> dict[str, list[str]]: (`dict[str, list[str]]`): A dictionary of each data type and any columns that don't match the required dtype and can't be converted to it successfully. + """ # Create a new mapping of the data's column names to the expected dtype # TODO: Consider if this should be a encoded in the metadata/plantdata object elsewhere @@ -901,11 +928,14 @@ def _validate_dtypes(self, category: str = "all") -> dict[str, list[str]]: zip( column_name_map[name][sub_name].values(), column_dtype_map[name][sub_name].values(), + strict=True, ) ) else: column_map[name] = dict( - zip(column_name_map[name].values(), column_dtype_map[name].values()) + zip( + column_name_map[name].values(), column_dtype_map[name].values(), strict=True + ) ) error_cols = {} @@ -917,12 +947,12 @@ def _validate_dtypes(self, category: str = "all") -> dict[str, list[str]]: if name == "reanalysis": if df is None: # If no reanalysis data, get the default key from ReanalysisMetaData - sub_name = [*column_map[name]][0] + sub_name = next(iter(column_map[name])) error_cols[f"{name}-{sub_name}"] = dtype_converter( df, column_types=column_map[name][sub_name] ) continue - for sub_name, df in df.items(): + for sub_name, df in df.items(): # noqa: B020 logger.info(f"Validating the data types in the {sub_name} {name} data") error_cols[f"{name}-{sub_name}"] = dtype_converter( df, column_types=column_map[name][sub_name] @@ -943,6 +973,7 @@ def _validate_frequency(self, category: str = "all") -> list[str]: Returns: list[str]: The list of data types that don't meet the required datetime frequency. + """ frequency_requirements = self.metadata.frequency_requirements(self.analysis_type) @@ -969,16 +1000,20 @@ def _validate_frequency(self, category: str = "all") -> list[str]: # If only checking one data type, then skip all others continue if name == "reanalysis": - for sub_name, freq in freq.items(): + for sub_name, freq in freq.items(): # noqa: B020 logger.info(f"Validating the frequency of the {sub_name} {name} data") - is_valid = frequency_validator(freq, frequency_requirements.get(name), True) - is_valid |= frequency_validator(freq, frequency_requirements.get(name), False) + is_valid = frequency_validator( + freq, frequency_requirements.get(name), exact=True + ) + is_valid |= frequency_validator( + freq, frequency_requirements.get(name), exact=False + ) if not is_valid: invalid_freq.update({f"{name}-{sub_name}": freq}) else: logger.info(f"Validating the frequency of the {name} data") - is_valid = frequency_validator(freq, frequency_requirements.get(name), True) - is_valid |= frequency_validator(freq, frequency_requirements.get(name), False) + is_valid = frequency_validator(freq, frequency_requirements.get(name), exact=True) + is_valid |= frequency_validator(freq, frequency_requirements.get(name), exact=False) if not is_valid: invalid_freq.update({name: freq}) @@ -987,7 +1022,7 @@ def _validate_frequency(self, category: str = "all") -> list[str]: @logged_method_call def validate(self, metadata: dict | str | Path | PlantMetaData | None = None) -> None: """Secondary method to validate the plant data objects after loading or changing - data with option to provide an updated `metadata` object/file as well + data with option to provide an updated `metadata` object/file as well. Args: metadata (Optional[dict]): Updated metadata object, dictionary, or file to @@ -996,6 +1031,7 @@ def validate(self, metadata: dict | str | Path | PlantMetaData | None = None) -> Raises: ValueError: Raised at the end if errors are caught in the validation steps. + """ logger.info("Post-intialization data validation") # Put the index columns back into the column space to ensure success of re-validation @@ -1088,6 +1124,7 @@ def parse_asset_geometry( Returns: None Sets the asset "geometry" column. + """ # Check for metadata inputs if utm_zone is None: @@ -1111,16 +1148,17 @@ def parse_asset_geometry( self.asset[self.metadata.asset.longitude].values, ) - self.asset["geometry"] = [Point(lat, lon) for lat, lon in zip(lats, lons)] + self.asset["geometry"] = [Point(lat, lon) for lat, lon in zip(lats, lons, strict=True)] @logged_method_call - def update_column_names(self, to_original: bool = False) -> None: + def update_column_names(self, *, to_original: bool = False) -> None: """Renames the columns of each dataframe to the be the keys from the `metadata.xx.col_map` that was passed during initialization. Args: to_original (bool, optional): An indicator to map the column names back to the originally passed values. Defaults to False. + """ meta = self.metadata reverse = not to_original # flip the boolean to correctly map between the col_map entries @@ -1153,6 +1191,7 @@ def update_column_names(self, to_original: bool = False) -> None: @logged_method_call def calculate_turbine_energy(self) -> None: + """Calculate the turbine energy generation from each time step from the power data.""" energy_col = self.metadata.scada.WTUR_SupWh power_col = self.metadata.scada.WTUR_W frequency = self.metadata.scada.frequency @@ -1180,6 +1219,7 @@ def turbine_df(self, turbine_id: str) -> pd.DataFrame: Returns: pd.DataFrame: The turbine-specific SCADA data frame. + """ if self.scada is None: raise AttributeError("This method can't be used unless `scada` data is provided.") @@ -1207,6 +1247,7 @@ def tower_df(self, tower_id: str) -> pd.DataFrame: Returns: pd.DataFrame: The met tower-specific data frame. + """ if self.tower is None: raise AttributeError("This method can't be used unless `tower` data is provided.") @@ -1230,6 +1271,7 @@ def calculate_asset_distance_matrix(self) -> pd.DataFrame: Returns: pd.DataFrame: Dataframe containing distances between each pair of assets + """ ix = self.asset.index.values distance = ( @@ -1254,7 +1296,7 @@ def calculate_asset_distance_matrix(self) -> pd.DataFrame: distance.loc[:, :] = distance_array self.asset_distance_matrix = distance - def turbine_distance_matrix(self, turbine_id: str = None) -> pd.DataFrame: + def turbine_distance_matrix(self, turbine_id: str | None = None) -> pd.DataFrame: """Returns the distances between all turbines in the plant with `np.inf` for the distance between a turbine and itself. @@ -1262,8 +1304,10 @@ def turbine_distance_matrix(self, turbine_id: str = None) -> pd.DataFrame: turbine_id (str, optional): Specific turbine ID for which the distances to other turbines are returned. If None, a matrix containing the distances between all pairs of turbines is returned. Defaults to None. + Returns: pd.DataFrame: Dataframe containing distances between each pair of turbines + """ if self.asset_distance_matrix.size == 0: self.calculate_asset_distance_matrix() @@ -1271,7 +1315,7 @@ def turbine_distance_matrix(self, turbine_id: str = None) -> pd.DataFrame: row_ix = self.turbine_ids if turbine_id is None else turbine_id return self.asset_distance_matrix.loc[row_ix, self.turbine_ids] - def tower_distance_matrix(self, tower_id: str = None) -> pd.DataFrame: + def tower_distance_matrix(self, tower_id: str | None = None) -> pd.DataFrame: """Returns the distances between all towers in the plant with `np.inf` for the distance between a tower and itself. @@ -1279,8 +1323,10 @@ def tower_distance_matrix(self, tower_id: str = None) -> pd.DataFrame: tower_id (str, optional): Specific tower ID for which the distances to other towers are returned. If None, a matrix containing the distances between all pairs of towers is returned. Defaults to None. + Returns: pd.DataFrame: Dataframe containing distances between each pair of towers + """ if self.asset_distance_matrix.size == 0: self.calculate_asset_distance_matrix() @@ -1296,6 +1342,7 @@ def calculate_asset_direction_matrix(self) -> pd.DataFrame: Returns: pd.DataFrame: Dataframe containing directions between each pair of assets (defined as the direction from the asset given by the row index to the asset given by the column index, relative to north) + """ ix = self.asset.index.values direction = ( @@ -1334,7 +1381,7 @@ def calculate_asset_direction_matrix(self) -> pd.DataFrame: direction.loc[:, :] = direction_array self.asset_direction_matrix = direction - def turbine_direction_matrix(self, turbine_id: str = None) -> pd.DataFrame: + def turbine_direction_matrix(self, turbine_id: str | None = None) -> pd.DataFrame: """Returns the directions between all turbines in the plant with `np.inf` for the direction between a turbine and itself. @@ -1342,10 +1389,12 @@ def turbine_direction_matrix(self, turbine_id: str = None) -> pd.DataFrame: turbine_id (str, optional): Specific turbine ID for which the directions to other turbines are returned. If None, a matrix containing the directions between all pairs of turbines is returned. Defaults to None. + Returns: pd.DataFrame: Dataframe containing directions between each pair of turbines (defined as the direction from the turbine given by the row index to the turbine given by the column index, relative to north) + """ if self.asset_direction_matrix.size == 0: self.calculate_asset_direction_matrix() @@ -1353,7 +1402,7 @@ def turbine_direction_matrix(self, turbine_id: str = None) -> pd.DataFrame: row_ix = self.turbine_ids if turbine_id is None else turbine_id return self.asset_direction_matrix.loc[row_ix, self.turbine_ids] - def tower_direction_matrix(self, tower_id: str = None) -> pd.DataFrame: + def tower_direction_matrix(self, tower_id: str | None = None) -> pd.DataFrame: """Returns the directions between all towers in the plant with `np.inf` for the direction between a tower and itself. @@ -1361,10 +1410,12 @@ def tower_direction_matrix(self, tower_id: str = None) -> pd.DataFrame: tower_id (str, optional): Specific tower ID for which the directions to other towers are returned. If None, a matrix containing the directions between all pairs of towers is returned. Defaults to None. + Returns: pd.DataFrame: Dataframe containing directions between each pair of towers (defined as the direction from the tower given by the row index to the tower given by the column index, relative to north) + """ if self.asset_direction_matrix.size == 0: self.calculate_asset_direction_matrix() @@ -1374,7 +1425,7 @@ def tower_direction_matrix(self, tower_id: str = None) -> pd.DataFrame: def calculate_asset_geometries(self) -> None: """Calculates the asset distances and parses the asset geometries. This is intended for use - during initialization and for when asset data is added after initialization + during initialization and for when asset data is added after initialization. """ if self.asset is not None: self.parse_asset_geometry() @@ -1384,8 +1435,7 @@ def calculate_asset_geometries(self) -> None: def get_freestream_turbines( self, wd: float, freestream_method: str = "sector", sector_width: float = 90.0 ): - """ - Returns a list of freestream (unwaked) turbines for a given wind direction. Freestream turbines can be + """Returns a list of freestream (unwaked) turbines for a given wind direction. Freestream turbines can be identified using different methods ("sector" or "IEC" methods). For the sector method, if there are any turbines upstream of a turbine within a fixed wind direction sector centered on the wind direction of interest, defined by the sector_width argument, the turbine is considered waked. The IEC method uses the freestream @@ -1399,8 +1449,10 @@ def get_freestream_turbines( interest used to determine whether a turbine is waked for the "sector" method (degrees). For a given turbine, if any other upstream turbines are located within the sector, then the turbine is considered waked. Defaults to 90 degrees. + Returns: list: List of freestream turbine asset IDs + """ turbine_direction_matrix = self.turbine_direction_matrix() @@ -1461,8 +1513,8 @@ def calculate_nearest_neighbor( Returns: None Creates the "nearest_turbine_id" and "nearest_tower_id" column in `asset`. - """ + """ # Get the valid IDs for both the turbines and towers ix_turb = self.turbine_ids if turbine_ids is None else np.array(turbine_ids) ix_tower = self.tower_ids if tower_ids is None else np.array(tower_ids) @@ -1491,6 +1543,7 @@ def nearest_turbine(self, asset_id: str) -> str: Returns: str: The turbine `asset_id` closest to the provided `asset_id`. + """ if "nearest_turbine_id" not in self.asset.columns: self.calculate_nearest_neighbor() @@ -1504,6 +1557,7 @@ def nearest_tower(self, asset_id: str) -> str: Returns: str: The tower `asset_id` closest to the provided `asset_id`. + """ if "nearest_tower_id" not in self.asset.columns: self.calculate_nearest_neighbor() @@ -1511,12 +1565,15 @@ def nearest_tower(self, asset_id: str) -> str: @classmethod def from_entr(cls, *args, **kwargs): + """Create a :py:class:`PlantData` object from an ENTR data base.""" try: from entr.plantdata import from_entr - except ModuleNotFoundError: - raise NotImplementedError( - "The entr python package was not found. Please install py-entr by visiting https://github.com/entralliance/py-entr and following the instructions." + except ModuleNotFoundError as e: + msg = ( + "The entr python package was not found. Please install py-entr by visiting" + " https://github.com/entralliance/py-entr and following the instructions." ) + raise NotImplementedError(msg) from e return from_entr(*args, **kwargs) @@ -1526,4 +1583,4 @@ def from_entr(cls, *args, **kwargs): # ********************************************************** # Add the method for fetching and attaching the EIA plant data to the project -setattr(PlantData, "attach_eia_data", attach_eia_data) +PlantData.attach_eia_data = attach_eia_data diff --git a/openoa/schema/metadata.py b/openoa/schema/metadata.py index 275be4436..3b0d45c55 100644 --- a/openoa/schema/metadata.py +++ b/openoa/schema/metadata.py @@ -1,3 +1,7 @@ +"""Provides the metadata dataclasses for configuring the user data mappings to OpenOA +naming conventions. +""" + from __future__ import annotations import re @@ -17,6 +21,7 @@ from openoa.logging import logging, logged_method_call + logger = logging.getLogger(__name__) warnings.filterwarnings("once", category=DeprecationWarning) @@ -192,6 +197,7 @@ def convert_frequency(offset: str) -> str: Args: offset (str): The alphanumeric offset string. Must be one of: "MS", "ME", "W", "D", "h", "min", "s", "ms", "us", "ns", "M", "H", "T", "S", "L", "U", or "N". + """ # Separate leading digits and the offset code offset_digits = re.findall(r"\d+", offset) @@ -210,8 +216,9 @@ def convert_frequency(offset: str) -> str: warnings.warn( f"Pandas 3.0 will deprecated the following codes, please use the following mapping {deprecated_offset_map}", DeprecationWarning, + stacklevel=2, ) - offset_str = deprecated_offset_map.get(offset_str, None) + offset_str = deprecated_offset_map.get(offset_str) elif offset_str not in _at_least_monthly: raise ValueError( @@ -238,6 +245,7 @@ def determine_analysis_requirements( Returns: dict | tuple[dict, dict]: The dictionary of column or frequency requirements, or if "both", then a tuple of each dictionary. + """ if isinstance(analysis_type, str): analysis_type = [analysis_type] @@ -267,7 +275,7 @@ def determine_analysis_requirements( frequency_requirements[name] = set(req) else: frequency_requirements[name] = reqs.intersection(req) - if which == "both": + if which == "both": # noqa: SIM116 return column_requirements, frequency_requirements elif which == "columns": return column_requirements @@ -282,20 +290,24 @@ class FromDictMixin: have a specific parameter definied. This allows passing of larger dictionaries to a data class without throwing an error. - Raises + Raises: ------ AttributeError Raised if the required class inputs are not provided. + """ @classmethod @logged_method_call def from_dict(cls, data: dict): """Maps a data dictionary to an `attrs`-defined class. + Args: data (dict): The data dictionary to be mapped. + Returns: (cls): An intialized object of the `attrs`-defined class (`cls`). + """ # Get all parameters from the input dictionary that map to the class initialization kwarg_names = [a.name for a in cls.__attrs_attrs__ if a.init] @@ -320,8 +332,7 @@ def from_dict(cls, data: dict): @define(auto_attribs=True) class ResetValuesMixin: - """ - A MixinClass that provides the methods to reset initialized or default values for analysis + """A MixinClass that provides the methods to reset initialized or default values for analysis parameters. """ @@ -331,6 +342,7 @@ def set_values(self, value_dict: dict): Args: value_dict (dict): The parameter names (keys) and their values (values) as a dictionary. + """ for name, value in value_dict.items(): logger.debug(f"{name} being set back to {value}") @@ -346,6 +358,7 @@ def reset_defaults(self, which: str | list[str] | tuple[str] | None = None): Raises: ValueError: Raised if any of :py:attr:`which` are not included in ``self.run_parameters``. + """ logger.info("Resetting run parameters back to the class defaults") # Define the analysis class run parameters @@ -381,10 +394,7 @@ def _make_single_repr(name: str, meta_class) -> str: axis=1, ) - if name == "ReanalysisMetaData": - repr = [] - else: - repr = ["-" * len(name), name, "-" * len(name) + "\n"] + repr = [] if name == "ReanalysisMetaData" else ["-" * len(name), name, "-" * len(name) + "\n"] if name != "AssetMetaData": repr.append("frequency\n--------") @@ -424,7 +434,7 @@ def _make_combined_repr(cls: PlantMetaData) -> str: @define(auto_attribs=True) -class SCADAMetaData(FromDictMixin): # noqa: F821 +class SCADAMetaData(FromDictMixin): """A metadata schematic to create the necessary column mappings and other validation components, or other data about the SCADA data, that will contribute to a larger plant metadata schema/routine. @@ -480,57 +490,59 @@ class SCADAMetaData(FromDictMixin): # noqa: F821 col_map: dict = field(init=False) col_map_reversed: dict = field(init=False) dtypes: dict = field( - default=dict( - time=np.datetime64, - asset_id=str, - WTUR_W=float, - WMET_HorWdSpd=float, - WMET_HorWdDir=float, - WMET_HorWdDirRel=float, - WTUR_TurSt=str, - WROT_BlPthAngVal=float, - WMET_EnvTmp=float, - WTUR_SupWh=float, - ), + default={ + "time": np.datetime64, + "asset_id": str, + "WTUR_W": float, + "WMET_HorWdSpd": float, + "WMET_HorWdDir": float, + "WMET_HorWdDirRel": float, + "WTUR_TurSt": str, + "WROT_BlPthAngVal": float, + "WMET_EnvTmp": float, + "WTUR_SupWh": float, + }, init=False, # don't allow for user input ) units: dict = field( - default=dict( - time="datetim64[ns]", - asset_id=None, - WTUR_W="kW", - WMET_HorWdSpd="m/s", - WMET_HorWdDir="deg", - WMET_HorWdDirRel="deg", - WTUR_TurSt=None, - WROT_BlPthAngVal="deg", - WMET_EnvTmp="C", - WTUR_SupWh="kWh", - ), + default={ + "time": "datetim64[ns]", + "asset_id": None, + "WTUR_W": "kW", + "WMET_HorWdSpd": "m/s", + "WMET_HorWdDir": "deg", + "WMET_HorWdDirRel": "deg", + "WTUR_TurSt": None, + "WROT_BlPthAngVal": "deg", + "WMET_EnvTmp": "C", + "WTUR_SupWh": "kWh", + }, init=False, # don't allow for user input ) def __attrs_post_init__(self) -> None: - self.col_map = dict( - time=self.time, - asset_id=self.asset_id, - WTUR_W=self.WTUR_W, - WMET_HorWdSpd=self.WMET_HorWdSpd, - WMET_HorWdDir=self.WMET_HorWdDir, - WMET_HorWdDirRel=self.WMET_HorWdDirRel, - WTUR_TurSt=self.WTUR_TurSt, - WROT_BlPthAngVal=self.WROT_BlPthAngVal, - WMET_EnvTmp=self.WMET_EnvTmp, - WTUR_SupWh=self.WTUR_SupWh, - ) + """Post initialization hook.""" + self.col_map = { + "time": self.time, + "asset_id": self.asset_id, + "WTUR_W": self.WTUR_W, + "WMET_HorWdSpd": self.WMET_HorWdSpd, + "WMET_HorWdDir": self.WMET_HorWdDir, + "WMET_HorWdDirRel": self.WMET_HorWdDirRel, + "WTUR_TurSt": self.WTUR_TurSt, + "WROT_BlPthAngVal": self.WROT_BlPthAngVal, + "WMET_EnvTmp": self.WMET_EnvTmp, + "WTUR_SupWh": self.WTUR_SupWh, + } self.col_map_reversed = {v: k for k, v in self.col_map.items()} def __repr__(self): + """Custom ``__repr__``. See :py:func:`_make_single_repr`.""" return _make_single_repr("SCADAMetaData", self) @define(auto_attribs=True) -class MeterMetaData(FromDictMixin): # noqa: F821 +class MeterMetaData(FromDictMixin): """A metadata schematic to create the necessary column mappings and other validation components, or other data about energy meter data, that will contribute to a larger plant metadata schema/routine. @@ -562,32 +574,28 @@ class MeterMetaData(FromDictMixin): # noqa: F821 name: str = field(default="meter", init=False) col_map: dict = field(init=False) dtypes: dict = field( - default=dict( - time=np.datetime64, - MMTR_SupWh=float, - ), + default={"time": np.datetime64, "MMTR_SupWh": float}, init=False, # don't allow for user input ) units: dict = field( - default=dict( - time="datetim64[ns]", - MMTR_SupWh="kWh", - ), + default={"time": "datetim64[ns]", "MMTR_SupWh": "kWh"}, init=False, # don't allow for user input ) def __attrs_post_init__(self) -> None: - self.col_map = dict( - time=self.time, - MMTR_SupWh=self.MMTR_SupWh, - ) + """Post initialization hook.""" + self.col_map = { + "time": self.time, + "MMTR_SupWh": self.MMTR_SupWh, + } def __repr__(self): + """Custom ``__repr__``. See :py:func:`_make_single_repr`.""" return _make_single_repr("MeterMetaData", self) @define(auto_attribs=True) -class TowerMetaData(FromDictMixin): # noqa: F821 +class TowerMetaData(FromDictMixin): """A metadata schematic to create the necessary column mappings and other validation components, or other data about meteorological tower (met tower) data, that will contribute to a larger plant metadata schema/routine. @@ -627,41 +635,43 @@ class TowerMetaData(FromDictMixin): # noqa: F821 name: str = field(default="tower", init=False) col_map: dict = field(init=False) dtypes: dict = field( - default=dict( - time=np.datetime64, - asset_id=str, - WMET_HorWdSpd=float, - WMET_HorWdDir=float, - WMET_EnvTmp=float, - ), + default={ + "time": np.datetime64, + "asset_id": str, + "WMET_HorWdSpd": float, + "WMET_HorWdDir": float, + "WMET_EnvTmp": float, + }, init=False, # don't allow for user input ) units: dict = field( - default=dict( - time="datetim64[ns]", - asset_id=None, - WMET_HorWdSpd="m/s", - WMET_HorWdDir="deg", - WMET_EnvTmp="C", - ), + default={ + "time": "datetim64[ns]", + "asset_id": None, + "WMET_HorWdSpd": "m/s", + "WMET_HorWdDir": "deg", + "WMET_EnvTmp": "C", + }, init=False, # don't allow for user input ) def __attrs_post_init__(self) -> None: - self.col_map = dict( - time=self.time, - asset_id=self.asset_id, - WMET_HorWdSpd=self.WMET_HorWdSpd, - WMET_HorWdDir=self.WMET_HorWdDir, - WMET_EnvTmp=self.WMET_EnvTmp, - ) + """Post initialization hook.""" + self.col_map = { + "time": self.time, + "asset_id": self.asset_id, + "WMET_HorWdSpd": self.WMET_HorWdSpd, + "WMET_HorWdDir": self.WMET_HorWdDir, + "WMET_EnvTmp": self.WMET_EnvTmp, + } def __repr__(self): + """Custom ``__repr__``. See :py:func:`_make_single_repr`.""" return _make_single_repr("TowerMetaData", self) @define(auto_attribs=True) -class StatusMetaData(FromDictMixin): # noqa: F821 +class StatusMetaData(FromDictMixin): """A metadata schematic to create the necessary column mappings and other validation components, or other data about the turbine status log data, that will contribute to a larger plant metadata schema/routine. @@ -701,41 +711,43 @@ class StatusMetaData(FromDictMixin): # noqa: F821 name: str = field(default="status", init=False) col_map: dict = field(init=False) dtypes: dict = field( - default=dict( - time=np.datetime64, - asset_id=str, - status_id=np.int64, - status_code=np.int64, - status_text=str, - ), + default={ + "time": np.datetime64, + "asset_id": str, + "status_id": np.int64, + "status_code": np.int64, + "status_text": str, + }, init=False, # don't allow for user input ) units: dict = field( - default=dict( - time="datetim64[ns]", - asset_id=None, - status_id=None, - status_code=None, - status_text=None, - ), + default={ + "time": "datetim64[ns]", + "asset_id": None, + "status_id": None, + "status_code": None, + "status_text": None, + }, init=False, # don't allow for user input ) def __attrs_post_init__(self) -> None: - self.col_map = dict( - time=self.time, - asset_id=self.asset_id, - status_id=self.status_id, - status_code=self.status_code, - status_text=self.status_text, - ) + """Post initialization hook.""" + self.col_map = { + "time": self.time, + "asset_id": self.asset_id, + "status_id": self.status_id, + "status_code": self.status_code, + "status_text": self.status_text, + } def __repr__(self): + """Custom ``__repr__``. See :py:func:`_make_single_repr`.""" return _make_single_repr("StatusMetaData", self) @define(auto_attribs=True) -class CurtailMetaData(FromDictMixin): # noqa: F821 +class CurtailMetaData(FromDictMixin): """A metadata schematic to create the necessary column mappings and other validation components, or other data about the plant curtailment data, that will contribute to a larger plant metadata schema/routine. @@ -769,35 +781,37 @@ class CurtailMetaData(FromDictMixin): # noqa: F821 name: str = field(default="curtail", init=False) col_map: dict = field(init=False) dtypes: dict = field( - default=dict( - time=np.datetime64, - IAVL_ExtPwrDnWh=float, - IAVL_DnWh=float, - ), + default={ + "time": np.datetime64, + "IAVL_ExtPwrDnWh": float, + "IAVL_DnWh": float, + }, init=False, # don't allow for user input ) units: dict = field( - default=dict( - time="datetim64[ns]", - IAVL_ExtPwrDnWh="kWh", - IAVL_DnWh="kWh", - ), + default={ + "time": "datetim64[ns]", + "IAVL_ExtPwrDnWh": "kWh", + "IAVL_DnWh": "kWh", + }, init=False, # don't allow for user input ) def __attrs_post_init__(self) -> None: - self.col_map = dict( - time=self.time, - IAVL_ExtPwrDnWh=self.IAVL_ExtPwrDnWh, - IAVL_DnWh=self.IAVL_DnWh, - ) + """Post initialization hook.""" + self.col_map = { + "time": self.time, + "IAVL_ExtPwrDnWh": self.IAVL_ExtPwrDnWh, + "IAVL_DnWh": self.IAVL_DnWh, + } def __repr__(self): + """Custom ``__repr__``. See :py:func:`_make_single_repr`.""" return _make_single_repr("CurtailMetaData", self) @define(auto_attribs=True) -class AssetMetaData(FromDictMixin): # noqa: F821 +class AssetMetaData(FromDictMixin): """A metadata schematic to create the necessary column mappings and other validation components, or other data about the site's asset metadata, that will contribute to a larger plant metadata schema/routine. @@ -817,6 +831,7 @@ class AssetMetaData(FromDictMixin): # noqa: F821 by default "elevation". This data should be of type: ``float``. type (str): The type of asset column in the asset metadata, by default "type". This data should be of type: ``str``. + """ # DataFrame columns @@ -834,54 +849,59 @@ class AssetMetaData(FromDictMixin): # noqa: F821 name: str = field(default="asset", init=False) col_map: dict = field(init=False) dtypes: dict = field( - default=dict( - asset_id=str, - latitude=float, - longitude=float, - rated_power=float, - hub_height=float, - rotor_diameter=float, - elevation=float, - type=str, - ), + default={ + "asset_id": str, + "latitude": float, + "longitude": float, + "rated_power": float, + "hub_height": float, + "rotor_diameter": float, + "elevation": float, + "type": str, + }, init=False, # don't allow for user input ) units: dict = field( - default=dict( - asset_id=None, - latitude="WGS84", - longitude="WGS84", - rated_power="kW", - hub_height="m", - rotor_diameter="m", - elevation="m", - type=None, - ), + default={ + "asset_id": None, + "latitude": "WGS84", + "longitude": "WGS84", + "rated_power": "kW", + "hub_height": "m", + "rotor_diameter": "m", + "elevation": "m", + "type": None, + }, init=False, # don't allow for user input ) def __attrs_post_init__(self) -> None: - self.col_map = dict( - asset_id=self.asset_id, - latitude=self.latitude, - longitude=self.longitude, - rated_power=self.rated_power, - hub_height=self.hub_height, - rotor_diameter=self.rotor_diameter, - elevation=self.elevation, - type=self.type, - ) + """Post initialization hook.""" + self.col_map = { + "asset_id": self.asset_id, + "latitude": self.latitude, + "longitude": self.longitude, + "rated_power": self.rated_power, + "hub_height": self.hub_height, + "rotor_diameter": self.rotor_diameter, + "elevation": self.elevation, + "type": self.type, + } def __repr__(self): + """Custom ``__repr__``. See :py:func:`_make_single_repr`.""" return _make_single_repr("AssetMetaData", self) def convert_reanalysis(value: dict[str, dict]): + """Convert nested reanalysis configuration dictionaries into a dictionary of + :py:attr:class`ReanalysisMetaData` objects. + """ return {k: ReanalysisMetaData.from_dict(v) for k, v in value.items()} @define(auto_attribs=True) -class ReanalysisMetaData(FromDictMixin): # noqa: F821 +class ReanalysisMetaData(FromDictMixin): """A metadata schematic for each of the reanalsis products to be used for operationa analyses to create the necessary column mappings and other validation components, or other data about the site's asset metadata, that will contribute to a larger plant metadata schema/routine. @@ -906,6 +926,7 @@ class ReanalysisMetaData(FromDictMixin): # noqa: F821 default "WMETR_EnvPres". frequency (:obj:`str`): The frequency of the timestamps in the :py:attr:`time` column, by default "10min". + """ time: str = field(default="time") @@ -925,50 +946,52 @@ class ReanalysisMetaData(FromDictMixin): # noqa: F821 name: str = field(default="reanalysis", init=False) col_map: dict = field(init=False) dtypes: dict = field( - default=dict( - time=np.datetime64, - WMETR_HorWdSpd=float, - WMETR_HorWdSpdU=float, - WMETR_HorWdSpdV=float, - WMETR_HorWdDir=float, - WMETR_EnvTmp=float, - WMETR_AirDen=float, - WMETR_EnvPres=float, - ), + default={ + "time": np.datetime64, + "WMETR_HorWdSpd": float, + "WMETR_HorWdSpdU": float, + "WMETR_HorWdSpdV": float, + "WMETR_HorWdDir": float, + "WMETR_EnvTmp": float, + "WMETR_AirDen": float, + "WMETR_EnvPres": float, + }, init=False, # don't allow for user input ) units: dict = field( - default=dict( - time="datetim64[ns]", - WMETR_HorWdSpd="m/s", - WMETR_HorWdSpdU="m/s", - WMETR_HorWdSpdV="m/s", - WMETR_HorWdDir="deg", - WMETR_EnvTmp="K", - WMETR_AirDen="kg/m^3", - WMETR_EnvPres="Pa", - ), + default={ + "time": "datetim64[ns]", + "WMETR_HorWdSpd": "m/s", + "WMETR_HorWdSpdU": "m/s", + "WMETR_HorWdSpdV": "m/s", + "WMETR_HorWdDir": "deg", + "WMETR_EnvTmp": "K", + "WMETR_AirDen": "kg/m^3", + "WMETR_EnvPres": "Pa", + }, init=False, # don't allow for user input ) def __attrs_post_init__(self) -> None: - self.col_map = dict( - time=self.time, - WMETR_HorWdSpd=self.WMETR_HorWdSpd, - WMETR_HorWdSpdU=self.WMETR_HorWdSpdU, - WMETR_HorWdSpdV=self.WMETR_HorWdSpdV, - WMETR_HorWdDir=self.WMETR_HorWdDir, - WMETR_EnvTmp=self.WMETR_EnvTmp, - WMETR_AirDen=self.WMETR_AirDen, - WMETR_EnvPres=self.WMETR_EnvPres, - ) + """Post initialization hook.""" + self.col_map = { + "time": self.time, + "WMETR_HorWdSpd": self.WMETR_HorWdSpd, + "WMETR_HorWdSpdU": self.WMETR_HorWdSpdU, + "WMETR_HorWdSpdV": self.WMETR_HorWdSpdV, + "WMETR_HorWdDir": self.WMETR_HorWdDir, + "WMETR_EnvTmp": self.WMETR_EnvTmp, + "WMETR_AirDen": self.WMETR_AirDen, + "WMETR_EnvPres": self.WMETR_EnvPres, + } def __repr__(self): + """Custom ``__repr__``. See :py:func:`_make_single_repr`.""" return _make_single_repr("ReanalysisMetaData", self) @define(auto_attribs=True) -class PlantMetaData(FromDictMixin): # noqa: F821 +class PlantMetaData(FromDictMixin): """Composese the metadata/validation requirements from each of the individual data types that can compose a `PlantData` object. @@ -1000,6 +1023,7 @@ class PlantMetaData(FromDictMixin): # noqa: F821 reanalysis type (as keys, such as "era5" or "merra2") and ``ReanalysisMetaData`` column mapping and frequency parameters for each type of reanalysis data provided. See ``ReanalysisMetaData`` for more details. + """ latitude: float = field(default=0, converter=float) @@ -1015,23 +1039,23 @@ class PlantMetaData(FromDictMixin): # noqa: F821 curtail: CurtailMetaData = field(default={}, converter=CurtailMetaData.from_dict) asset: AssetMetaData = field(default={}, converter=AssetMetaData.from_dict) reanalysis: dict[str, ReanalysisMetaData] = field( - default={"product": {}}, converter=convert_reanalysis # noqa: F821 - ) # noqa: F821 + default={"product": {}}, converter=convert_reanalysis + ) @property def column_map(self) -> dict[str, dict]: """Provides the column mapping for all of the available data types with the name of each data type as the key and the dictionary mapping as the values. """ - values = dict( - scada=self.scada.col_map, - meter=self.meter.col_map, - tower=self.tower.col_map, - status=self.status.col_map, - asset=self.asset.col_map, - curtail=self.curtail.col_map, - reanalysis={}, - ) + values = { + "scada": self.scada.col_map, + "meter": self.meter.col_map, + "tower": self.tower.col_map, + "status": self.status.col_map, + "asset": self.asset.col_map, + "curtail": self.curtail.col_map, + "reanalysis": {}, + } if self.reanalysis != {}: values["reanalysis"] = {k: v.col_map for k, v in self.reanalysis.items()} return values @@ -1041,15 +1065,15 @@ def dtype_map(self) -> dict[str, dict]: """Provides the column dtype matching for all of the available data types with the name of each data type as the keys, and the column dtype mapping as values. """ - types = dict( - scada=self.scada.dtypes, - meter=self.meter.dtypes, - tower=self.tower.dtypes, - status=self.status.dtypes, - asset=self.asset.dtypes, - curtail=self.curtail.dtypes, - reanalysis={}, - ) + types = { + "scada": self.scada.dtypes, + "meter": self.meter.dtypes, + "tower": self.tower.dtypes, + "status": self.status.dtypes, + "asset": self.asset.dtypes, + "curtail": self.curtail.dtypes, + "reanalysis": {}, + } if self.reanalysis != {}: types["reanalysis"] = {k: v.dtypes for k, v in self.reanalysis.items()} return types @@ -1060,6 +1084,7 @@ def coordinates(self) -> tuple[float, float]: Returns: tuple[float, float]: The (latitude, longitude) pair + """ return self.latitude, self.longitude @@ -1075,12 +1100,13 @@ def from_json(cls, metadata_file: str | Path) -> PlantMetaData: Returns: PlantMetaData + """ metadata_file = Path(metadata_file).resolve() if not metadata_file.is_file(): raise FileExistsError(f"Input JSON file: {metadata_file} is an invalid input.") - with open(metadata_file) as f: + with metadata_file.open() as f: return cls.from_dict(json.load(f)) @classmethod @@ -1095,12 +1121,13 @@ def from_yaml(cls, metadata_file: str | Path) -> PlantMetaData: Returns: PlantMetaData + """ metadata_file = Path(metadata_file).resolve() if not metadata_file.is_file(): raise FileExistsError(f"Input YAML file: {metadata_file} is an invalid input.") - with open(metadata_file) as f: + with metadata_file.open() as f: return cls.from_dict(yaml.safe_load(f)) @classmethod @@ -1108,7 +1135,7 @@ def load(cls, data: str | Path | dict | PlantMetaData) -> PlantMetaData: """Loads the metadata from either a dictionary or file such as a JSON or YAML file. Args: - metadata_file (`str | Path | dict`): Either a pre-loaded dictionary or + data (`str | Path | dict`): Either a pre-loaded dictionary or the full path and file name of the JSON or YAML file. Raises: @@ -1117,6 +1144,7 @@ def load(cls, data: str | Path | dict | PlantMetaData) -> PlantMetaData: Returns: PlantMetaData + """ if isinstance(data, PlantMetaData): return data @@ -1148,6 +1176,7 @@ def frequency_requirements(self, analysis_types: list[str | None]) -> dict[str, Returns: dict[str, set[str]]: The dictionary of data type name and valid frequencies for the datetime stamps. + """ if "all" in analysis_types: requirements = deepcopy(ANALYSIS_REQUIREMENTS) @@ -1175,4 +1204,5 @@ def frequency_requirements(self, analysis_types: list[str | None]) -> dict[str, return frequency def __repr__(self): + """Custom repr. See :py:func:`_make_combined_repr`.""" return _make_combined_repr(self) diff --git a/openoa/schema/schema.py b/openoa/schema/schema.py index 7aca35f74..40540c726 100644 --- a/openoa/schema/schema.py +++ b/openoa/schema/schema.py @@ -1,4 +1,4 @@ -"""Methods to generate YAML and JSON schema files""" +"""Methods to generate YAML and JSON schema files.""" from __future__ import annotations @@ -21,6 +21,7 @@ determine_analysis_requirements, ) + HERE = Path(__file__).resolve().parent meta_class_map = { @@ -43,12 +44,11 @@ def _attrs_meta_filter(inst: Attribute, value: Any) -> bool: Returns: bool: False, if should not be serialized, and True, if it should be serialized. + """ if inst.name in ("col_map", "name", "col_map_reversed"): return False - if inst is None or value is None: - return False - return True + return not (inst is None or value is None) def _attrs_meta_serializer(inst: type, field: Attribute, value: Any) -> Any: @@ -61,6 +61,7 @@ def _attrs_meta_serializer(inst: type, field: Attribute, value: Any) -> Any: Returns: Any: Reformatted data. + """ if field is None: return value @@ -75,13 +76,14 @@ def create_schema() -> dict: Returns: dict: The compiled metadata dictionary specifying the required data definitions. + """ schema = {name: {} for name in meta_class_map} for name, meta in meta_class_map.items(): meta_dict = asdict( meta(), filter=_attrs_meta_filter, value_serializer=_attrs_meta_serializer ) - for key, value in meta_dict.items(): + for key in meta_dict: if key in ("dtypes", "units"): continue if key == "frequency": @@ -100,6 +102,7 @@ def create_analysis_schema(analysis_types: str | list[str]) -> dict: Returns: dict: The compiled metadata dictionary specifying the required data definitions. + """ schema = create_schema() schema_copy = deepcopy(schema) @@ -133,52 +136,52 @@ def create_analysis_schema(analysis_types: str | list[str]) -> dict: base_yaw_misalignment_schema = create_analysis_schema("StaticYawMisalignment") # Save the analysis schemass - with open(HERE / "full_schema.yml", "w") as f: + with (HERE / "full_schema.yml", "w").open() as f: yaml.dump(full_schema, f, default_flow_style=False, sort_keys=False) - with open(HERE / "full_schema.json", "w") as f: + with (HERE / "full_schema.json", "w").open() as f: json.dump(full_schema, f, sort_keys=False, indent=2) - with open(HERE / "base_monte_carlo_aep_schema.yml", "w") as f: + with (HERE / "base_monte_carlo_aep_schema.yml", "w").open() as f: yaml.dump(base_mc_aep_schema, f, default_flow_style=False, sort_keys=False) - with open(HERE / "base_monte_carlo_aep_schema.json", "w") as f: + with (HERE / "base_monte_carlo_aep_schema.json", "w").open() as f: json.dump(base_mc_aep_schema, f, sort_keys=False, indent=2) - with open(HERE / "temperature_monte_carlo_aep_schema.yml", "w") as f: + with (HERE / "temperature_monte_carlo_aep_schema.yml", "w").open() as f: yaml.dump(temp_mc_aep_schema, f, default_flow_style=False, sort_keys=False) - with open(HERE / "temperature_monte_carlo_aep_schema.json", "w") as f: + with (HERE / "temperature_monte_carlo_aep_schema.json", "w").open() as f: json.dump(temp_mc_aep_schema, f, sort_keys=False, indent=2) - with open(HERE / "temperature_wind_direction_monte_carlo_aep_schema.yml", "w") as f: + with (HERE / "temperature_wind_direction_monte_carlo_aep_schema.yml", "w").open() as f: yaml.dump(temp_wd_mc_aep_schema, f, default_flow_style=False, sort_keys=False) - with open(HERE / "temperature_wind_direction_monte_carlo_aep_schema.json", "w") as f: + with (HERE / "temperature_wind_direction_monte_carlo_aep_schema.json", "w").open() as f: json.dump(temp_wd_mc_aep_schema, f, sort_keys=False, indent=2) - with open(HERE / "wind_direction_monte_carlo_aep_schema.yml", "w") as f: + with (HERE / "wind_direction_monte_carlo_aep_schema.yml", "w").open() as f: yaml.dump(wd_mc_aep_schema, f, default_flow_style=False, sort_keys=False) - with open(HERE / "wind_direction_monte_carlo_aep_schema.json", "w") as f: + with (HERE / "wind_direction_monte_carlo_aep_schema.json", "w").open() as f: json.dump(wd_mc_aep_schema, f, sort_keys=False, indent=2) - with open(HERE / "scada_wake_losses_schema.yml", "w") as f: + with (HERE / "scada_wake_losses_schema.yml", "w").open() as f: yaml.dump(scada_wake_schema, f, default_flow_style=False, sort_keys=False) - with open(HERE / "scada_wake_losses_schema.json", "w") as f: + with (HERE / "scada_wake_losses_schema.json", "w").open() as f: json.dump(scada_wake_schema, f, sort_keys=False, indent=2) - with open(HERE / "tower_wake_losses_schema.yml", "w") as f: + with (HERE / "tower_wake_losses_schema.yml", "w").open() as f: yaml.dump(tower_wake_schema, f, default_flow_style=False, sort_keys=False) - with open(HERE / "tower_wake_losses_schema.json", "w") as f: + with (HERE / "tower_wake_losses_schema.json", "w").open() as f: json.dump(tower_wake_schema, f, sort_keys=False, indent=2) - with open(HERE / "base_tie_schema.yml", "w") as f: + with (HERE / "base_tie_schema.yml", "w").open() as f: yaml.dump(base_tie_schema, f, default_flow_style=False, sort_keys=False) - with open(HERE / "base_tie_schema.json", "w") as f: + with (HERE / "base_tie_schema.json", "w").open() as f: json.dump(base_tie_schema, f, sort_keys=False, indent=2) - with open(HERE / "base_electrical_losses_schema.yml", "w") as f: + with (HERE / "base_electrical_losses_schema.yml", "w").open() as f: yaml.dump(base_electric_schema, f, default_flow_style=False, sort_keys=False) - with open(HERE / "base_electrical_losses_schema.json", "w") as f: + with (HERE / "base_electrical_losses_schema.json", "w").open() as f: json.dump(base_electric_schema, f, sort_keys=False, indent=2) - with open(HERE / "base_yaw_misalignmental_losses_schema.yml", "w") as f: + with (HERE / "base_yaw_misalignmental_losses_schema.yml", "w").open() as f: yaml.dump(base_yaw_misalignment_schema, f, default_flow_style=False, sort_keys=False) - with open(HERE / "base_yaw_misalignmental_losses_schema.json", "w") as f: + with (HERE / "base_yaw_misalignmental_losses_schema.json", "w").open() as f: json.dump(base_yaw_misalignment_schema, f, sort_keys=False, indent=2) diff --git a/openoa/utils/_converters.py b/openoa/utils/_converters.py index bcf426f84..df9edb259 100644 --- a/openoa/utils/_converters.py +++ b/openoa/utils/_converters.py @@ -1,15 +1,16 @@ -""" -This is module for common data conversion and checking methods and decorators that are used +"""This is module for common data conversion and checking methods and decorators that are used throughout the utils subpackage. """ from __future__ import annotations +import contextlib from math import ceil -from typing import Any, Type, Callable +from typing import Any from inspect import getfullargspec from functools import wraps from itertools import filterfalse +from collections.abc import Callable import pandas as pd @@ -24,6 +25,7 @@ def _list_of_len(x: list, length: int) -> list: Returns: list: A list of length :py:attr:`length` with repeating elements of :py:attr:`x`. + """ if (actual := len(x)) == length: return x @@ -40,6 +42,7 @@ def _check_cols_in_df(data, *args): Raises: ValueError: Raised if one of the values provided to :py:attr:`args` is not column or ``None``. + """ if any(isinstance(arg, pd.Series) for arg in args): raise TypeError( @@ -61,9 +64,10 @@ def _get_arguments(args: list, kwargs: dict, arg_ix_list: list[int], data_cols: Returns: ``list``: A list of the extracted values or None if it does not exist. + """ arg_list = [] - for ix, name in zip(arg_ix_list, data_cols): + for ix, name in zip(arg_ix_list, data_cols, strict=True): try: arg_list.append(args[ix]) except IndexError: @@ -91,16 +95,15 @@ def _update_arguments( Returns: tuple[list, dict]: _description_ + """ - for ix, name, new in zip(arg_ix_list, data_cols, arg_list): + for ix, name, new in zip(arg_ix_list, data_cols, arg_list, strict=True): try: args[ix] = new except IndexError: - try: + # No need to pass through a non-existent input that is defaulted to None + with contextlib.suppress(KeyError): kwargs[name] = new - except KeyError: - # No need to pass through a non-existent input that is defaulted to None - pass return args, kwargs @@ -114,6 +117,7 @@ def convert_args_to_lists(length: int, *args) -> list[list]: Returns: list[list]: A list of lists of length :py:attr:`length` for each argument passed. + """ return [a if isinstance(a, list) else [a] * length for a in args] @@ -134,13 +138,16 @@ def df_to_series(data: pd.DataFrame, *args: str) -> tuple[pd.Series, ...]: Returns: tuple[pandas.Series, ...]: A pandas `Series` for each of the column names passed in `args` + """ if len(args) == 0: raise ValueError("No column names provided to args for conversion to Series objects.") series_args = [isinstance(arg, pd.Series) for arg in args] if data is None: - if all(el or isinstance(arg, type(None)) for el, arg in zip(series_args, args)): + if all( + el or isinstance(arg, type(None)) for el, arg in zip(series_args, args, strict=True) + ): return args raise ValueError("No input provided to `data`; cannot convert args to Series.") if not isinstance(data, pd.DataFrame): @@ -171,6 +178,7 @@ def multiple_df_to_single_df(*args: pd.DataFrame, align_col: str | None = None) Returns: pd.DataFrame: _description_ + """ if not all(isinstance(el, pd.DataFrame) for el in args): raise TypeError("At least one of the provided values was not a pandas DataFrame") @@ -184,7 +192,9 @@ def multiple_df_to_single_df(*args: pd.DataFrame, align_col: str | None = None) return pd.concat(args, join="outer", axis=1) -def series_to_df(*args: pd.Series, names: list[str] = None) -> tuple[pd.DataFrame, list[str | int]]: +def series_to_df( + *args: pd.Series, names: list[str] | None = None +) -> tuple[pd.DataFrame, list[str | int]]: """Convert a dynamic number of pandas ``Series`` to a single pandas ``DataFrame`` by concatenating with an outer join, so the any missing values being filled with a NaN value, and each argument becomes a column of the resulting ``DataFrame``. @@ -197,6 +207,7 @@ def series_to_df(*args: pd.Series, names: list[str] = None) -> tuple[pd.DataFram Returns: ``tuple[pandas.DataFrame, list[str | int, ...]]``: A single data structure combining all the passed arguments, and the `name` associated with each passed `Series`. + """ if not all(isinstance(el, pd.Series) for el in args): raise TypeError("At least one of the provided values was not a pandas Series") @@ -204,8 +215,10 @@ def series_to_df(*args: pd.Series, names: list[str] = None) -> tuple[pd.DataFram # Rename the series to the name of the method argument if it doesn't already have name if names is None: names = [None] * len(args) - names = [name if el.name is None else el.name for el, name in zip(args, names)] - args = [el.rename(name) if el.name is None else el for el, name in zip(args, names)] + names = [name if el.name is None else el.name for el, name in zip(args, names, strict=True)] + args = [ + el.rename(name) if el.name is None else el for el, name in zip(args, names, strict=True) + ] args = [el.to_frame() for el in args] if len(args) > 1: @@ -213,7 +226,7 @@ def series_to_df(*args: pd.Series, names: list[str] = None) -> tuple[pd.DataFram return args[0], names -def series_method(data_cols: list[str] = None): +def series_method(data_cols: list[str] | None = None): """Wrapper method for methods that operate on pandas ``Series``, and not ``DataFrame``s that allows the passing of column names that are potentially contained in a pandas ``DataFrame`` to be pulled out as separate pandas ``Series`` objects to be passed back to the method. This is a convenience @@ -224,10 +237,11 @@ def series_method(data_cols: list[str] = None): data_cols (list[str], optional): The names of the method arguments that should be converted from ``str`` to pandas ``Series`` when ``data`` is provided as a pandas ``DataFrame`` to the focal method. Defaults to None. + """ def decorator(func: Callable): - """Gathes the arg indices from :py:attr:`data_cols` to be used in ``wrapper``.""" + """Gathers the arg indices from :py:attr:`data_cols` to be used in ``wrapper``.""" argspec = getfullargspec(func) arg_ix_list = [] if data_cols is not None: @@ -235,10 +249,9 @@ def decorator(func: Callable): @wraps(func) def wrapper(*args: Any, **kwargs: Any): - if no_df := (df := kwargs.get("data", None)) is None: - if arg_ix_list == []: - # Let the original method handle the provided arguments if unconfigured - return func(*args, *kwargs) + if (no_df := (df := kwargs.get("data")) is None) and arg_ix_list == []: + # Let the original method handle the provided arguments if unconfigured + return func(*args, *kwargs) args = list(args) @@ -256,7 +269,7 @@ def wrapper(*args: Any, **kwargs: Any): return decorator -def dataframe_method(data_cols: list[str] = None): +def dataframe_method(data_cols: list[str] | None = None): """Wrapper method for methods that operate on a pandas ``DataFrame``, and not ``Series`` that allows the passing of the ``Series``, so that they can be combined in a ``DataFrame`` and passed back to the method. This is a convenience wrapper that reduces the amount of boilerplate required to enable @@ -266,6 +279,7 @@ def dataframe_method(data_cols: list[str] = None): data_cols (list[str], optional): The names of the method arguments that should be converted from a pandas ``Series`` to ``str``, along with the creation of the :py:attr:`data` keyword argument when the column data is passed as a ``Series``. Defaults to None. + """ def decorator(func: Callable): @@ -279,7 +293,7 @@ def decorator(func: Callable): def wrapper(*args: Any, **kwargs: Any): args = list(args) arg_list = _get_arguments(args, kwargs, arg_ix_list, data_cols) - if (df := kwargs.get("data", None)) is not None: + if (df := kwargs.get("data")) is not None: # If a DataFrame is provided and the wrapper is unconfigured, then pass straight to the function if arg_ix_list == []: return func(*args, *kwargs) diff --git a/openoa/utils/downloader.py b/openoa/utils/downloader.py index 2e71f6cf6..91b62aab1 100644 --- a/openoa/utils/downloader.py +++ b/openoa/utils/downloader.py @@ -1,5 +1,4 @@ -""" -This module provides functions for downloading files, including arbitrary files, files from Zenodo, +"""This module provides functions for downloading files, including arbitrary files, files from Zenodo, and reanalysis data. It contains functions for downloading long-term historical atmospheric data from the MERRA2 and @@ -33,7 +32,6 @@ from __future__ import annotations -import os import re import hashlib import datetime @@ -51,6 +49,7 @@ from openoa.utils import met_data_processing as met from openoa.logging import logging + logger = logging.getLogger() @@ -58,8 +57,7 @@ def download_file(url: str, outfile: str | Path) -> None: - """ - Download a file from the web, based on its url, and save to the outfile. + """Download a file from the web, based on its url, and save to the outfile. Args: url(:obj:`str`): Url of data to download. @@ -68,8 +66,8 @@ def download_file(url: str, outfile: str | Path) -> None: Raises: HTTPError: If unable to access url. Exception: If the request failed for another reason. - """ + """ outfile = Path(outfile).resolve() result = requests.get(url, stream=True) @@ -97,8 +95,7 @@ def download_file(url: str, outfile: str | Path) -> None: def download_zenodo_data(record_id: int, outfile_path: str | Path) -> None: - """ - Download data from Zenodo based on the Zenodo record_id. + """Download data from Zenodo based on the Zenodo record_id. The following files will be saved to the asset data folder: @@ -110,7 +107,6 @@ def download_zenodo_data(record_id: int, outfile_path: str | Path) -> None: outfile_path(:obj:`str` | :obj:`Path`): Path to save files to. """ - url_zenodo = r"https://zenodo.org/api/records/" r = requests.get(f"{url_zenodo}{record_id}") @@ -186,11 +182,10 @@ def get_era5_monthly( save_pathname: str | Path, save_filename: str, start_date: str = "2000-01", - end_date: str = None, + end_date: str | None = None, ) -> pd.DataFrame: - """ - Get ERA5 data directly from the CDS service. This requires registration on the CDS service. - See registration details at: https://cds.climate.copernicus.eu/how-to-api + """Get ERA5 data directly from the CDS service. This requires registration on the CDS service. + See registration details at: https://cds.climate.copernicus.eu/how-to-api. This function returns monthly ERA5 data from the "ERA5 monthly averaged data on single levels from 1959 to present" dataset at the nearest grid point to the provided coordinates. See @@ -225,8 +220,8 @@ def get_era5_monthly( Raises: ValueError: If the start_date is greater than the end_date. Exception: If unable to connect to the cdsapi client. - """ + """ logger.info("Please note access to ERA5 data requires registration") logger.info("Please see: https://cds.climate.copernicus.eu/api-how-to") @@ -341,7 +336,7 @@ def get_era5_monthly( # delete downloaded NetCDF files for file in save_pathname.glob(f"{save_filename}*.nc"): - os.remove(file) + file.unlink() return df @@ -352,13 +347,13 @@ def get_era5_hourly( save_pathname: str | Path, save_filename: str, start_date: str = "2000-01", - end_date: str = None, + end_date: str | None = None, + *, calc_derived_vars: bool = False, ) -> pd.DataFrame: - """ - Get ERA5 data directly from the CDS service. This requires registration on the CDS service and + """Get ERA5 data directly from the CDS service. This requires registration on the CDS service and an API key to be saved. See registration and API setup details at: - https://cds.climate.copernicus.eu/how-to-api + https://cds.climate.copernicus.eu/how-to-api. This function returns hourly ERA5 data from the "ERA5 hourly data on single levels from 1940 to present" dataset at the nearest grid point to the provided coordinates. See further details @@ -397,8 +392,8 @@ def get_era5_hourly( Raises: ValueError: If the start_date is greater than the end_date. - """ + """ logger.info("Please note access to ERA5 data requires registration") logger.info("Please see: https://cds.climate.copernicus.eu/api-how-to") @@ -406,7 +401,7 @@ def get_era5_hourly( try: c = cdsapi.Client() except SSLError: - print("Skipping certificate verification") + print("Skipping certificate verification") # noqa: T201 c = cdsapi.Client(verify=False) # verification error for self-signed certificate # create save_pathname if it does not exist @@ -457,7 +452,6 @@ def get_era5_hourly( "month": None, "day": [f"{i:02d}" for i in range(1, 32)], "time": [f"{i:02d}:00" for i in range(24)], - "product_type": "reanalysis", "area": [ lat_nearest, lon_nearest, @@ -533,7 +527,7 @@ def get_era5_hourly( # delete downloaded NetCDF files for file in save_pathname.glob(f"{save_filename}*.nc"): - os.remove(file) + file.unlink() return df @@ -544,10 +538,9 @@ def get_merra2_monthly( save_pathname: str | Path, save_filename: str, start_date: str = "2000-01", - end_date: str = None, + end_date: str | None = None, ) -> pd.DataFrame: - """ - Get MERRA2 data directly from the NASA GES DISC service, which requires registration on the + """Get MERRA2 data directly from the NASA GES DISC service, which requires registration on the GES DISC service. See: https://disc.gsfc.nasa.gov/information/documents?title=Data%20Access#python-requests. This function returns monthly MERRA2 data from the "M2IMNXLFO" dataset at the nearest grid @@ -580,8 +573,8 @@ def get_merra2_monthly( Raises: ValueError: If the start_year is greater than the end_year. - """ + """ logger.info("Please note access to MERRA2 data requires registration") logger.info( "Please see: https://disc.gsfc.nasa.gov/information/documents?title=Data%20Access#python-requests" @@ -691,7 +684,7 @@ def get_merra2_monthly( # delete downloaded NetCDF files for file in save_pathname.glob(f"{save_filename}*.nc"): - os.remove(file) + file.unlink() return df @@ -702,11 +695,11 @@ def get_merra2_hourly( save_pathname: str | Path, save_filename: str, start_date: str = "2000-01", - end_date: str = None, + end_date: str | None = None, + *, calc_derived_vars: bool = False, ) -> pd.DataFrame: - """ - Get MERRA2 data directly from the NASA GES DISC service, which requires registration on the + """Get MERRA2 data directly from the NASA GES DISC service, which requires registration on the GES DISC service. See: https://disc.gsfc.nasa.gov/information/documents?title=Data%20Access#python-requests. This function returns hourly MERRA2 data from the "M2T1NXSLV" dataset at the nearest grid point @@ -744,8 +737,8 @@ def get_merra2_hourly( Raises: ValueError: If the start_date is greater than the end_date. - """ + """ logger.info("Please note access to MERRA2 data requires registration") logger.info( "Please see: https://disc.gsfc.nasa.gov/information/documents?title=Data%20Access#python-requests" @@ -877,6 +870,6 @@ def get_merra2_hourly( # delete downloaded NetCDF files for file in save_pathname.glob(f"{save_filename}*.nc"): - os.remove(file) + file.unlink() return df diff --git a/openoa/utils/filters.py b/openoa/utils/filters.py index 3d025e4d8..ceea563d7 100644 --- a/openoa/utils/filters.py +++ b/openoa/utils/filters.py @@ -1,5 +1,4 @@ -""" -This module provides functions for flagging pandas data series based on a range of criteria. The functions are largely +"""This module provides functions for flagging pandas data series based on a range of criteria. The functions are largely intended for application in wind plant operational energy analysis, particularly wind speed vs. power curves. """ @@ -30,7 +29,7 @@ def range_flag( data (:obj:`pandas.Series` | `pandas.DataFrame`): data frame containing the column to be flagged; can either be a ``pandas.Series`` or ``pandas.DataFrame``. If a ``pandas.DataFrame``, a list of threshold values and columns (if checking a subset of the columns) must be provided. - col (:obj:`list[str]`): column(s) in :pyattr:`data` to be flagged, by default None. Only + col (:obj:`list[str]`): column(s) in :py:attr:`data` to be flagged, by default None. Only required when the `data` is a ``pandas.DataFrame`` and a subset of the columns will be checked. Must be the same length as :py:attr:`lower` and :py:attr:`upper`. lower (:obj:`float` | `list[float]`): lower threshold (inclusive) for each element of :py:attr:`data`, @@ -45,6 +44,7 @@ def range_flag( Returns: :obj:`pandas.Series` | `pandas.DataFrame`: Series or DataFrame (depending on :py:attr:`data` type) with boolean entries. + """ # Prepare the inputs to be standardized for use with DataFrames if to_series := isinstance(data, pd.Series): @@ -85,6 +85,7 @@ def unresponsive_flag( Returns: :obj:`pandas.Series` | `pandas.DataFrame`: Series or DataFrame (depending on ``data`` type) with boolean entries. + """ # Prepare the inputs to be standardized for use with DataFrames if to_series := isinstance(data, pd.Series): @@ -136,6 +137,7 @@ def std_range_flag( Returns: :obj:`pandas.Series` | `pandas.DataFrame`: Series or DataFrame (depending on :py:attr:`data` type) with boolean entries. + """ # Prepare the inputs to be standardized for use with DataFrames if to_series := isinstance(data, pd.Series): @@ -183,6 +185,7 @@ def window_range_flag( Returns: :obj:`pandas.Series`: Series with boolean entries. + """ flag = window_col.between(window_start, window_end) & ~value_col.between(value_min, value_max) return flag @@ -195,15 +198,15 @@ def bin_filter( bin_width: float, threshold: float = 2, center_type: str = "mean", - bin_min: float = None, - bin_max: float = None, + bin_min: float | None = None, + bin_max: float | None = None, threshold_type: str = "std", direction: str = "all", data: pd.DataFrame = None, ): """Flag time stamps for which data in `value_col` when binned by data in `bin_col` into bins of width `bin_width` are outside the `threhsold` bin. The `center_type` of each bin can be either the - median or mean, and flagging can be applied directionally (i.e. above or below the center, or both) + median or mean, and flagging can be applied directionally (i.e. above or below the center, or both). Args: bin_col(:obj:`pandas.Series` | `str`): The Series or column in :py:attr:`data` to be used for binning. @@ -221,6 +224,7 @@ def bin_filter( Returns: :obj:`pandas.Series(bool)`: Array-like object with boolean entries. + """ if center_type not in ("mean", "median"): raise ValueError("Incorrect `center_type` specified; must be one of 'mean' or 'median'.") @@ -311,6 +315,7 @@ def cluster_mahalanobis_2d( Returns: :obj:`pandas.Series(bool)`: Array-like object with boolean entries. + """ data = data.loc[:, [data_col1, data_col2]].copy() kmeans = KMeans(n_clusters=n_clusters).fit(data) @@ -333,7 +338,8 @@ def cluster_mahalanobis_2d( # Compute mahalnobis distance of each point in cluster mahalanobis_dist = cluster.apply( - lambda r: sp.spatial.distance.mahalanobis(r.values, centroid, invcovmx), axis=1 + lambda r: sp.spatial.distance.mahalanobis(r.values, centroid, invcovmx), # noqa: B023 + axis=1, ) # Flag data outside the distance threshold diff --git a/openoa/utils/imputing.py b/openoa/utils/imputing.py index 03124ee7b..81a1b8d5f 100644 --- a/openoa/utils/imputing.py +++ b/openoa/utils/imputing.py @@ -1,12 +1,9 @@ -""" -This module provides methods for filling in null data with interpolated (imputed) values. -""" +"""This module provides methods for filling in null data with interpolated (imputed) values.""" from copy import deepcopy import numpy as np import pandas as pd -from tqdm import tqdm from numpy.polynomial import Polynomial @@ -22,6 +19,7 @@ def asset_correlation_matrix(data: pd.DataFrame, value_col: str) -> pd.DataFrame Returns: :obj:`pandas.DataFrame`: Correlation matrix with as index and column names + """ corr_df = data.loc[:, [value_col]].unstack().corr(min_periods=2) corr_df = corr_df.droplevel(0).droplevel(0, axis=1) # drop the added axes @@ -36,7 +34,7 @@ def impute_data( reference_col: str, target_data: pd.DataFrame = None, reference_data: pd.DataFrame = None, - align_col: str = None, + align_col: str | None = None, method: str = "linear", degree: int = 1, data: pd.DataFrame = None, @@ -61,9 +59,14 @@ def impute_data( target_data(:obj:`pandas.DataFrame`): the ``DataFrame`` with NaN data to be imputed. reference_data(:obj:`pandas.DataFrame`): the ``DataFrame`` to be used in imputation align_col(:obj:`str`): the name of the column that to join :py:attr:`target_data` and :py:attr:`reference_data`. + degree (:obj:`int`): Degree of the polynomial fit when :py:attr:`method` is "polynomial". + Defaults to 1. + method (:obj:`str`): Imputation method. Only 1-d polynomials are currently allowed, so must + be one of "linear" or "polynomial". Defaults to "linear". Returns: :obj:`pandas.Series`: Copy of target_data_col series with NaN occurrences imputed where possible. + """ final_col_name = deepcopy(target_col) if data is None: diff --git a/openoa/utils/machine_learning_setup.py b/openoa/utils/machine_learning_setup.py index 63af6103d..1afb04388 100644 --- a/openoa/utils/machine_learning_setup.py +++ b/openoa/utils/machine_learning_setup.py @@ -1,5 +1,4 @@ -""" -This module is a library of machine learning algorithms and associated hyperparameter ranges +"""This module is a library of machine learning algorithms and associated hyperparameter ranges suitable for wind energy analysis. This module allow for simple implementation of hyperparameter optimization and the application of the best hyperparameter combinations for use in the predictive model. @@ -72,6 +71,7 @@ def _algorithm_map( Returns: GAM | ExtraTreesRegressor | GradientBoostingRegressor: The actual model. + """ if abbreviation == "etr": return ExtraTreesRegressor() @@ -98,6 +98,7 @@ class MachineLearningSetup: :py:class:`pygam.GAM` model, respectively. params(:obj:`dict`): Custom hyperparameter settings to be used for the passed :py:attr:`algorithm`. + """ algorithm: str = field(converter=(str.lower, _algorithm_map)) @@ -111,17 +112,8 @@ class MachineLearningSetup: opt_model: Any = field(init=False) def __attrs_post_init__(self): - """ - Initialize the hyperparameter ranges and scorer object - """ - if isinstance(self.algorithm, ExtraTreesRegressor): - self.hyper_range = { - "max_depth": [4, 8, 12, 16, 20], - "min_samples_split": np.arange(2, 11), - "min_samples_leaf": np.arange(1, 11), - "n_estimators": np.arange(10, 801, 40), - } - elif isinstance(self.algorithm, GradientBoostingRegressor): + """Initialize the hyperparameter ranges and scorer object.""" + if isinstance(self.algorithm, (ExtraTreesRegressor, GradientBoostingRegressor)): self.hyper_range = { "max_depth": [4, 8, 12, 16, 20], "min_samples_split": np.arange(2, 11), @@ -137,8 +129,7 @@ def __attrs_post_init__(self): self.my_scorer = make_scorer(r2_score, greater_is_better=True) def hyper_report(self, results: dict, n_top: int = 5) -> None: - """ - Output hyperparameter optimization results into terminal window in order of mean validation score. + """Output hyperparameter optimization results into terminal window in order of mean validation score. Args: results(:obj:'dict'): Dictionary containg cross-validation results @@ -146,37 +137,37 @@ def hyper_report(self, results: dict, n_top: int = 5) -> None: Returns: (none): Top :py:param:`n_top` results are printed. - """ + """ # Loop through cross validation results and output to terminal for i in range(1, n_top + 1): candidates = np.flatnonzero(results["rank_test_score"] == i) for candidate in candidates: - print(f"Model with rank: {i}\n") + print(f"Model with rank: {i}\n") # noqa: T201 message = ( f"Mean validation score: {results['mean_test_score'][candidate]:.3f} " f"(std: {results['std_test_score'][candidate]:.3f})\n" ) - print(message) - print(f"Parameters: {results['params'][candidate]}\n") - print("") + print(message) # noqa: T201 + print(f"Parameters: {results['params'][candidate]}\n") # noqa: T201 + print("") # noqa: T201 def hyper_optimize( self, X: np.ndarray | pd.DataFrame, y: np.ndarray | pd.Series, - cv: sklearn.model_selection._split = KFold(n_splits=5), + cv: sklearn.model_selection._split | None = None, n_iter_search: int = 20, - report: bool = True, verbose: int = 0, n_jobs: int | None = None, + *, + report: bool = True, ) -> None: - """ - Optimize hyperparameters through cross-validation + """Optimize hyperparameters through cross-validation. Args: X(:obj:'numpy.ndarray` | `pandas.DataFrame`): The inputs or features. - Y(:obj:'numpy.ndarray` | `pandas.Series`): The target or to-be-predicted data. + y(:obj:'numpy.ndarray` | `pandas.Series`): The target or to-be-predicted data. cv(:obj:'sklearn.model_selection._split'): The train/test splitting method. Defaults to :py:class:`KFold(n_splits=5)`. n_iter_search(:obj:'int'): The number of random hyperparmeter samples to use. Defaults @@ -195,7 +186,10 @@ def hyper_optimize( Returns: (none) + """ + if cv is None: + cv = KFold(n_splits=5) # Setup randomized cross-validated grid search self.random_search = RandomizedSearchCV( self.algorithm, diff --git a/openoa/utils/met_data_processing.py b/openoa/utils/met_data_processing.py index a546689fe..310dc9d8f 100644 --- a/openoa/utils/met_data_processing.py +++ b/openoa/utils/met_data_processing.py @@ -1,6 +1,4 @@ -""" -This module provides methods for processing meteorological data. -""" +"""This module provides methods for processing meteorological data.""" from __future__ import annotations @@ -12,21 +10,23 @@ from openoa.utils._converters import df_to_series, series_method + # Define constants used in some of the methods R = 287.058 # Gas constant for dry air, units of J/kg/K Rw = 461.5 # Gas constant of water vapour, unit J/kg/K def wrap_180(x: float | np.ndarray | pd.Series | pd.DataFrame): - """ - Converts an angle, an array of angles, or a pandas Series or DataFrame of angles in degrees to + """Converts an angle, an array of angles, or a pandas Series or DataFrame of angles in degrees to the range -180 to +180 degrees. Args: x (float | np.ndarray | pd.Series | pd.DataFrame): Input angle(s) (degrees) + Returns: float | np.ndarray: The input angle(s) converted to the range -180 to +180 degrees, returned as a float or numpy array (degrees) + """ input_type = type(x) if (input_type == pd.core.series.Series) | (input_type == pd.core.frame.DataFrame): @@ -41,9 +41,8 @@ def wrap_180(x: float | np.ndarray | pd.Series | pd.DataFrame): def circular_mean(x: pd.DataFrame | pd.Series | np.ndarray, axis: int = 0): - """ - Compute circular mean of wind direction data for a pandas Series or 1-dimensional numpy array, - or along any dimension of a multi-dimensional pandas DataFrame or numpy array + """Compute circular mean of wind direction data for a pandas Series or 1-dimensional numpy array, + or along any dimension of a multi-dimensional pandas DataFrame or numpy array. Args: x(pd.DataFrame | pd.Series | np.ndarray): A pandas DataFrame or Series, or a numpy array @@ -54,6 +53,7 @@ def circular_mean(x: pd.DataFrame | pd.Series | np.ndarray, axis: int = 0): Returns: pd.Series | float | np.ndarray: The circular mean of the wind directions along the specified axis between 0 and 360 degrees (degrees). + """ if axis >= x.ndim: raise ValueError("The axis argument cannot be greater than the dimension of the data (x).") @@ -93,6 +93,7 @@ def compute_wind_speed( Returns: :obj:`pandas.Series` | :obj:`numpy.ndarray`: wind speed, in m/s. + """ return np.sqrt(u**2 + v**2) @@ -101,7 +102,7 @@ def compute_wind_speed( def compute_wind_direction( u: pd.Series | str, v: pd.Series | str, data: pd.DataFrame = None ) -> pd.Series: - """Compute wind direction given u and v wind vector components + """Compute wind direction given u and v wind vector components. Args: u(:obj:`pandas.Series` | `str`): A pandas ``Series`` of the zonal component of the wind, @@ -112,6 +113,7 @@ def compute_wind_direction( Returns: :obj:`pandas.Series`: wind direction; units of degrees + """ wd = 180 + np.arctan2(u, v) * 180 / np.pi # Calculate wind direction in degrees return pd.Series(np.where(wd != 360, wd, 0)) @@ -121,7 +123,7 @@ def compute_wind_direction( def compute_u_v_components( wind_speed: pd.Series | str, wind_dir: pd.Series | str, data: pd.DataFrame = None ) -> pd.Series: - """Compute vector components of the horizontal wind given wind speed and direction + """Compute vector components of the horizontal wind given wind speed and direction. Args: wind_speed(:obj:`pandas.Series` | `str`): A pandas ``Series`` of the horizontal wind speed, in @@ -138,6 +140,7 @@ def compute_u_v_components( (tuple): u(pandas.Series): the zonal component of the wind; units of m/s. v(pandas.Series): the meridional component of the wind; units of m/s + """ if np.any(wind_speed < 0): raise ValueError("Negative values exist in the `wind_speed` data.") @@ -157,8 +160,7 @@ def compute_air_density( humi_col: pd.Series | str = None, data: pd.DataFrame = None, ): - """ - Calculate air density from the ideal gas law based on the definition provided by IEC 61400-12 + """Calculate air density from the ideal gas law based on the definition provided by IEC 61400-12 given pressure, temperature and relative humidity. This function assumes temperature and pressure are reported in standard units of measurement @@ -182,6 +184,7 @@ def compute_air_density( Returns: :obj:`pandas.Series`: Rho, calcualted air density; units of kg/m3 + """ if data is not None: temp_col, pres_col, humi_col = df_to_series(data, temp_col, pres_col, humi_col) @@ -210,8 +213,7 @@ def pressure_vertical_extrapolation( z1: pd.Series | str, data: pd.DataFrame = None, ) -> pd.Series: - """ - Extrapolate pressure from height z0 to height z1 given the average temperature in the layer. + """Extrapolate pressure from height z0 to height z1 given the average temperature in the layer. The hydostatic equation is used to peform the extrapolation. Args: @@ -231,6 +233,7 @@ def pressure_vertical_extrapolation( Returns: :obj:`pandas.Series`: :py:attr:`p1`, extrapolated pressure at :py:attr:`z1`, in Pascals + """ if np.any(p0 < 0): raise ValueError("Negative values exist in the `p0` data.") @@ -244,8 +247,7 @@ def pressure_vertical_extrapolation( def air_density_adjusted_wind_speed( wind_col: pd.Series | str, density_col: pd.Series | str, data: pd.DataFrame = None ) -> pd.Series: - """ - Apply air density correction to wind speed measurements following IEC-61400-12-1 standard + """Apply air density correction to wind speed measurements following IEC-61400-12-1 standard. Args: wind_col(:obj:`pandas.Series` | `str`): A pandas `Series` containing the wind speed data, @@ -257,6 +259,7 @@ def air_density_adjusted_wind_speed( Returns: :obj:`pandas.Series`: density-adjusted wind speeds, in m/s + """ return wind_col * np.power(density_col / density_col.mean(), 1.0 / 3) @@ -265,8 +268,7 @@ def air_density_adjusted_wind_speed( def compute_turbulence_intensity( mean_col: pd.Series | str, std_col: pd.Series | str, data: pd.DataFrame = None ) -> pd.Series: - """ - Compute turbulence intensity + """Compute turbulence intensity. Args: mean_col(:obj:`pandas.Series` | `str`): A pandas ``Series`` containing the wind speed mean @@ -278,6 +280,7 @@ def compute_turbulence_intensity( Returns: :obj:`pd.Series`: turbulence intensity, (unitless ratio) + """ if data is not None: mean_col, std_col = df_to_series(data, mean_col, std_col) @@ -285,10 +288,9 @@ def compute_turbulence_intensity( def compute_shear( - data: pd.DataFrame, ws_heights: dict[str, float], return_reference_values: bool = False + data: pd.DataFrame, ws_heights: dict[str, float], *, return_reference_values: bool = False ) -> pd.Series | tuple[pd.Series, float, pd.Series]: - """ - Computes shear coefficient between wind speed measurements using the power law. + """Computes shear coefficient between wind speed measurements using the power law. The shear coefficient is obtained by evaluating the expression for an OLS regression coefficient. Args: @@ -306,8 +308,8 @@ def compute_shear( :py:attr:`return_reference_values` is False, return just the shear coefficient (unitless), else return the shear coefficent (unitless), reference height (m), and reference wind speed (m/s). - """ + """ # Extract the wind speed columns from `data` and create "u" 2-D array; where element # [i,j] is the wind speed measurement at the ith timestep and jth sensor height u: np.ndarray = np.column_stack(df_to_series(data, *ws_heights)) @@ -358,8 +360,7 @@ def compute_shear( def extrapolate_windspeed( v1: pd.Series | str, z1: float, z2: float, shear: pd.Series | str, data: pd.DataFrame = None ): - """ - Extrapolates wind speed vertically using the Power Law. + """Extrapolates wind speed vertically using the Power Law. Args: v1(:obj: `pandas.Series` | `float` | `str`): A pandas ``Series`` of the wind @@ -372,6 +373,7 @@ def extrapolate_windspeed( Returns: :obj: (`pandas.Series` | `numpy.array` | `float`): Wind speed extrapolated to target height. + """ return v1 * (z2 / z1) ** shear @@ -384,8 +386,7 @@ def compute_veer( height_b: float, data: pd.DataFrame = None, ): - """ - Compute veer between wind direction measurements + """Compute veer between wind direction measurements. Args: wind_a(:obj:`pandas.Series` | `str`): A pandas ``Series`` containing the wind direction mean @@ -399,6 +400,7 @@ def compute_veer( Returns: veer(:obj:`array`): veer (deg/m) + """ # Calculate wind direction change delta_dir = wind_b - wind_a diff --git a/openoa/utils/metadata_fetch.py b/openoa/utils/metadata_fetch.py index 8a052778e..438dda428 100644 --- a/openoa/utils/metadata_fetch.py +++ b/openoa/utils/metadata_fetch.py @@ -1,6 +1,4 @@ -""" -This module fetches metadata of wind farms -""" +"""This module fetches metadata of wind farms.""" from __future__ import annotations @@ -13,6 +11,7 @@ from openoa.utils import unit_conversion + if TYPE_CHECKING: from openoa import PlantData @@ -26,10 +25,10 @@ def fetch_eia( wind_file: str | Path, wind_sheet: str | Path, ): - """ - Read in EIA data of wind farm of interest: - - from EIA API for monthly productions, return monthly net energy generation time series - - from local Excel files for wind farm metadata, return dictionary of metadata + """Read in EIA data of wind farm of interest. + + - from EIA API for monthly productions, return monthly net energy generation time series + - from local Excel files for wind farm metadata, return dictionary of metadata Args: api_key(:obj:`str`): 32-character user-specific API key, obtained from EIA. @@ -108,7 +107,7 @@ def meta_dic_fn(metafile: str | Path, sheet: str, var_list: list[str]): api = eia.API(api_key) # get data from EIA - series_search_m = api.data_by_series(series="ELEC.PLANT.GEN.%s-ALL-ALL.M" % plant_id) + series_search_m = api.data_by_series(series=f"ELEC.PLANT.GEN.{plant_id}-ALL-ALL.M") eia_monthly = pd.DataFrame(series_search_m) # net monthly energy generation of wind farm in MWh eia_monthly.columns = ["eia_monthly_mwh"] # rename column eia_monthly = eia_monthly.set_index( @@ -128,8 +127,7 @@ def attach_eia_data( wind_file: str | Path, wind_sheet: str | Path, ): - """ - Assign EIA meta data to PlantData object, which is by default an empty dictionary. + """Assign EIA meta data to PlantData object, which is by default an empty dictionary. Args: project(:obj:`PlantData`): PlantData object for a particular project @@ -145,6 +143,7 @@ def attach_eia_data( Returns: (None) + """ project.eia["api_key"] = api_key project.eia["data_dir"] = file_path diff --git a/openoa/utils/plot.py b/openoa/utils/plot.py index 9d604ef32..39d523cb4 100644 --- a/openoa/utils/plot.py +++ b/openoa/utils/plot.py @@ -1,7 +1,4 @@ -""" -This module provides helpful functions for creating various plots - -""" +"""This module provides helpful functions for creating various plots.""" from __future__ import annotations @@ -18,7 +15,6 @@ from bokeh.plotting import figure from matplotlib.ticker import StrMethodFormatter -from openoa import PlantData NDArrayFloat = npt.NDArray[np.float64] @@ -97,7 +93,7 @@ def map_wgs84_to_cartesian( def luminance(rgb: tuple[int, int, int]): - """Calculates the brightness of an rgb 255 color. See https://en.wikipedia.org/wiki/Relative_luminance + """Calculates the brightness of an rgb 255 color. See https://en.wikipedia.org/wiki/Relative_luminance. Args: rgb(:obj:`tuple`): Tuple of red, gree, and blue values in the range of 0-255. @@ -117,14 +113,13 @@ def luminance(rgb: tuple[int, int, int]): 0.21243529411764706 """ - luminance = (0.2126 * rgb[0] + 0.7152 * rgb[1] + 0.0722 * rgb[2]) / 255 return luminance def color_to_rgb(color: str | tuple[int, int, int]): - """Converts named colors, hex and normalised RGB to 255 RGB values + """Converts named colors, hex and normalised RGB to 255 RGB values. Args: color(:obj:`color`): RGB, HEX or named color. @@ -144,9 +139,9 @@ def color_to_rgb(color: str | tuple[int, int, int]): >>> color_to_rgb("#ff00ff") (255,0,255) - """ - if isinstance(color, tuple): + """ + if isinstance(color, tuple): # noqa: SIM102 if max(color) > 1: color = tuple([i / 255 for i in color]) @@ -163,8 +158,8 @@ def plot_windfarm( plot_width=800, plot_height=800, marker_size=14, - figure_kwargs={}, - marker_kwargs={}, + figure_kwargs: dict | None = None, + marker_kwargs: dict | None = None, ): """Plot the windfarm spatially on a map using the Bokeh plotting libaray. @@ -199,7 +194,12 @@ def plot_windfarm( # Create the bokeh wind farm plot show(plot_windfarm(project.asset, tile_name="ESRI", plot_width=600, plot_height=600)) + """ + if figure_kwargs is None: + figure_kwargs = {} + if marker_kwargs is None: + marker_kwargs = {} # See https://wiki.openstreetmap.org/wiki/Tile_servers for various tile services MAP_TILES = { @@ -216,7 +216,7 @@ def plot_windfarm( asset_df["x"], asset_df["y"] = TRANSFORM_4326_TO_3857.transform( asset_df["latitude"], asset_df["longitude"] ) - asset_df["coordinates"] = tuple(zip(asset_df["latitude"], asset_df["longitude"])) + asset_df["coordinates"] = tuple(zip(asset_df["latitude"], asset_df["longitude"], strict=True)) # Define default and then update figure and marker options based on kwargs figure_options = { @@ -249,7 +249,7 @@ def plot_windfarm( else: color_palette = viridis(len(set(asset_df[color_grouping]))) - color_mapping = dict(zip(set(asset_df[color_grouping]), color_palette)) + color_mapping = dict(zip(set(asset_df[color_grouping]), color_palette, strict=False)) asset_df["auto_fill_color"] = asset_df[color_grouping].map(color_mapping) asset_df["auto_fill_color"] = asset_df["auto_fill_color"].apply(color_to_rgb) asset_df["auto_line_color"] = [ @@ -299,9 +299,10 @@ def plot_by_id( ylim: tuple[float, float] = (None, None), xlabel: str | None = None, ylabel: str | None = None, - return_fig: bool = False, figure_kwargs: dict | None = None, plot_kwargs: dict | None = None, + *, + return_fig: bool = False, ) -> None: """Function to plot any two fields against each other in a dataframe with unique plots for each asset_id. @@ -329,6 +330,7 @@ def plot_by_id( Returns: (:obj:`None`) + """ # Operate on a totally new copy of the data so that transofrmations don't carry through df = df.copy() @@ -367,7 +369,7 @@ def plot_by_id( # Create the plot fig, axes_list = plt.subplots(num_rows, max_cols, sharex=True, sharey=True, **figure_kwargs) - for i, (t_id, ax) in enumerate(zip(id_arrary, axes_list.flatten())): + for i, (t_id, ax) in enumerate(zip(id_arrary, axes_list.flatten(), strict=False)): scada = df.loc[t_id] ax.scatter(scada[x_axis], scada[y_axis], **plot_kwargs) @@ -395,16 +397,19 @@ def plot_by_id( return fig, axes_list -def column_histograms(df: pd.DataFrame, columns: list = None, return_fig: bool = False): +def column_histograms(df: pd.DataFrame, columns: list | None = None, *, return_fig: bool = False): """Produces a histogram plot for each numeric column in :py:attr:`df`. Args: df(:obj:`pd.DataFrame`): The dataframe for plotting. + columns (:obj:`list[str]`): A list of string column names contained in :py:attr:`df` that + should have histograms plotted. return_fig(:obj:`bool`): Indicator for if the figure and axes objects should be returned, by default False. Returns: (None) + """ df = df.select_dtypes((int, float)).copy() columns = df.columns.tolist() if columns is None else columns @@ -413,7 +418,7 @@ def column_histograms(df: pd.DataFrame, columns: list = None, return_fig: bool = num_rows = int(np.ceil(num_cols / max_cols)) fig, axes_list = plt.subplots(num_rows, max_cols, figsize=(15, num_rows * 5)) - for i, (col, ax) in enumerate(zip(columns, axes_list.flatten())): + for i, (col, ax) in enumerate(zip(columns, axes_list.flatten(), strict=False)): data = df.loc[:, col].dropna().values ax.hist(data, 40) ax.set_title(col) @@ -441,12 +446,13 @@ def plot_power_curve( flag_labels: tuple[str, str] = ("Flagged Readings", "Power Curve"), xlim: tuple[float, float] = (None, None), ylim: tuple[float, float] = (None, None), - legend: bool = False, - return_fig: bool = False, figure_kwargs: dict | None = None, legend_kwargs: dict | None = None, scatter_kwargs: dict | None = None, -) -> None | tuple[plt.Figure, plt.Axes]: + *, + legend: bool = False, + return_fig: bool = False, +) -> tuple[plt.Figure, plt.Axes] | None: """Plots the individual points on a power curve, with an optional :py:attr:`flag` filtering for singling out readings in the figure. If `flag` is all false values then no overlaid flagge scatter points will be created. @@ -478,6 +484,7 @@ def plot_power_curve( Returns: None | tuple[plt.Figure, plt.Axes]: _description_ + """ if figure_kwargs is None: figure_kwargs = {} @@ -521,21 +528,22 @@ def plot_monthly_reanalysis_windspeed( data: dict[str, pd.DataFrame], windspeed_col: str, plant_por: tuple[datetime.datetime, datetime.datetime], - normalize: bool = True, xlim: tuple[datetime.datetime, datetime.datetime] = (None, None), ylim: tuple[float, float] = (None, None), - return_fig: bool = False, figure_kwargs: dict | None = None, plot_kwargs: dict | None = None, legend_kwargs: dict | None = None, -) -> None | tuple[plt.Figure, plt.Axes]: + *, + normalize: bool = True, + return_fig: bool = False, +) -> tuple[plt.Figure, plt.Axes] | None: """Make a plot of the normalized annual average wind speeds from reanalysis data to show general trends for each, and highlighting the period of record for the plant data. Args: data(:obj:`dict[pandas.DataFrame]`): The dictionary of reanalysis dataframes. windspeed_col(:obj:`str`): The name of the column for the windspeed data to be plot. - plot_por(:obj:`tuple[datetime.datetime, datetime.datetime]`): The start and end datetimes + plant_por(:obj:`tuple[datetime.datetime, datetime.datetime]`): The start and end datetimes for a plant's period of record (POR). normalize(:obj:`bool`): Indicator of if the windspeeds shoudld be normalized (True), or not (False). Defaults to True. @@ -554,6 +562,7 @@ def plot_monthly_reanalysis_windspeed( Returns: None | tuple[matplotlib.pyplot.Figure, matplotlib.pyplot.Axes]: If :py:attr:`return_fig` is True, then the figure and axes objects are returned for further tinkering/saving. + """ # Define parameters needed for plotting min_val, max_val = (np.inf, -np.inf) if ylim == (None, None) else ylim @@ -618,13 +627,13 @@ def plot_plant_energy_losses_timeseries( xlim: tuple[datetime.datetime, datetime.datetime] = (None, None), ylim_energy: tuple[float, float] = (None, None), ylim_loss: tuple[float, float] = (None, None), - return_fig: bool = False, figure_kwargs: dict | None = None, plot_kwargs: dict | None = None, legend_kwargs: dict | None = None, + *, + return_fig: bool = False, ): - """ - Plot timeseries of energy, and the loss categories of interest. + """Plot timeseries of energy, and the loss categories of interest. Args: data(:obj:`pandas.DataFrame`): A pandas DataFrame containing energy production and losses. @@ -650,6 +659,7 @@ def plot_plant_energy_losses_timeseries( None | tuple[matplotlib.pyplot.Figure, tuple[matplotlib.pyplot.Axes, matplotlib.pyplot.Axes]]: If :py:attr:`return_fig` is True, then the figure and axes objects are returned for further tinkering/saving. + """ if figure_kwargs is None: figure_kwargs = {} @@ -671,7 +681,7 @@ def plot_plant_energy_losses_timeseries( ax1.set_ylabel(energy_label) # Joint availability and curtailment plot - for col, label in zip(loss_cols, loss_labels): + for col, label in zip(loss_cols, loss_labels, strict=True): ax2.plot(data[col] * 100, ".-", label=label, **plot_kwargs) ax2.set_xlabel("Year") ax2.set_ylabel("Loss (%)") @@ -694,19 +704,19 @@ def plot_distributions( data: pd.DataFrame, which: list[str], xlabels: list[str], - xlim: tuple[tuple[float, float], ...] = None, - ylim: tuple[tuple[float, float], ...] = None, - return_fig: bool = False, + xlim: tuple[tuple[float, float], ...] | None = None, + ylim: tuple[tuple[float, float], ...] | None = None, figure_kwargs: dict | None = None, plot_kwargs: dict | None = None, annotate_kwargs: dict | None = None, title: str | None = None, -) -> None | tuple[plt.Figure, plt.Axes]: - """ - Plot a distribution of AEP values from the Monte-Carlo OA method + *, + return_fig: bool = False, +) -> tuple[plt.Figure, plt.Axes] | None: + """Plot a distribution of AEP values from the Monte-Carlo OA method. Args: - aep(:obj:`pandas.DataFrame`): The pandas DataFrame of results data. + data(:obj:`pandas.DataFrame`): The pandas DataFrame of results data. which:(:obj:`list[str]`): The list of columns in data that should have their distributions plot. xlabels:(obj:`list[str]`): The list of x-axis labels xlim(:obj:`tuple[tuple[float, float], ...]`, optional): A tuple of tuples (or None) @@ -727,6 +737,7 @@ def plot_distributions( Returns: None | tuple[matplotlib.pyplot.Figure, matplotlib.pyplot.Axes]: If :py:attr:`return_fig` is True, then the figure and axes objects are returned for further tinkering/saving. + """ if figure_kwargs is None: figure_kwargs = {} @@ -750,9 +761,11 @@ def plot_distributions( figure_kwargs.setdefault("figsize", (14, 12)) figure_kwargs.setdefault("dpi", 200) fig = plt.figure(**figure_kwargs) - axes = fig.subplots(2, 2, gridspec_kw=dict(wspace=0.1, hspace=0.2)) + axes = fig.subplots(2, 2, gridspec_kw={"wspace": 0.1, "hspace": 0.2}) - for ax, col, label, _xlim, _ylim in zip(axes.flatten(), which, xlabels, xlim, ylim): + for ax, col, label, _xlim, _ylim in zip( + axes.flatten(), which, xlabels, xlim, ylim, strict=True + ): vals = data[col].values u_vals = vals.mean() ax.hist(vals, 40, density=1, **plot_kwargs) @@ -802,6 +815,7 @@ def _generate_swarm_values(y, n_bins=None, width: float = 0.5): Returns: :obj:`numpy.ndarray` An array of x-coordinates to plot as a scatter against :py:attr:`y`. + """ if n_bins is None: n_bins = y.size // 6 @@ -829,7 +843,7 @@ def _generate_swarm_values(y, n_bins=None, width: float = 0.5): # Assign the x indices in alternating fashion for each bin to ensure the x values are roughly symmetric dx = 1 / (n_max // 2) - for i, vals in zip(ix_bin_groups, y_bin_groups): + for i, vals in zip(ix_bin_groups, y_bin_groups, strict=True): if len(i) > 1: j = len(i) % 2 i = i[np.argsort(vals)] @@ -847,15 +861,16 @@ def plot_boxplot( xlabel: str, ylabel: str, ylim: tuple[float | None, float | None] = (None, None), - with_points: bool = False, points_label: str | None = None, - return_fig: bool = False, figure_kwargs: dict | None = None, plot_kwargs_box: dict | None = None, plot_kwargs_points: dict | None = None, legend_kwargs: dict | None = None, -) -> None | tuple[plt.Figure, plt.Axes]: - """Plot box plots of AEP results sliced by a specified Monte Carlo parameter + *, + with_points: bool = False, + return_fig: bool = False, +) -> tuple[plt.Figure, plt.Axes] | None: + """Plot box plots of AEP results sliced by a specified Monte Carlo parameter. Args: x(:obj:`pandas.Series`): The data that splits the results in y. @@ -882,6 +897,7 @@ def plot_boxplot( None | tuple[matplotlib.pyplot.Figure, matplotlib.pyplot.Axes, dict]: If :py:attr:`return_fig` is True, then the figure object, axes object, and a dictionary of the boxplot objects are returned for further tinkering/saving. + """ if figure_kwargs is None: figure_kwargs = {} @@ -911,7 +927,7 @@ def plot_boxplot( plot_kwargs_points.setdefault("facecolor", "none") plot_kwargs_points.setdefault("edgecolor", "green") plot_kwargs_points.setdefault("alpha", 0.5) - for x_start, (_y, width) in enumerate(zip(y_groups, widths)): + for x_start, (_y, width) in enumerate(zip(y_groups, widths, strict=True)): _x = _generate_swarm_values(_y, width=width * 0.9) + x_start + 1 label = points_label if x_start == width.size - 1 else None ax.scatter(_x, _y, zorder=0, label=label, **plot_kwargs_points) @@ -939,12 +955,12 @@ def plot_waterfall( index: list[str], ylabel: str | None = None, ylim: tuple[float, float] = (None, None), - return_fig: bool = False, plot_kwargs: dict | None = None, figure_kwargs: dict | None = None, -) -> None | tuple: - """ - Produce a waterfall plot showing the progression from the EYA estimates to the calculated OA + *, + return_fig: bool = False, +) -> tuple | None: + """Produce a waterfall plot showing the progression from the EYA estimates to the calculated OA estimates of AEP. Args: @@ -965,6 +981,7 @@ def plot_waterfall( Returns: None | tuple[plt.Figure, plt.Axes]: If :py:attr:`return_fig`, then return the figure and axes objects in addition to showing the plot. + """ if figure_kwargs is None: figure_kwargs = {} @@ -994,7 +1011,7 @@ def plot_waterfall( x = np.arange(plot_data.shape[0]) ax.bar(x, plot_data.amount, bottom=bottom, **plot_kwargs) ax.hlines( - bottom[1:-1].tolist() + [total], + [*bottom[1:-1].tolist(), total], xmin=x[:-1] - width / 2.0, xmax=x[:-1] + 1 + width / 2.0, colors="tab:orange", @@ -1003,7 +1020,7 @@ def plot_waterfall( # Add the annotations above/below each bar with a +/- label on difference for each category offset_pos = plot_data.amount.max() * 0.05 offset_neg = plot_data.amount.max() * 0.09 - for i, (y, diff) in enumerate(zip(bottom.values, plot_data.amount.values)): + for i, (y, diff) in enumerate(zip(bottom.values, plot_data.amount.values, strict=True)): if i in (0, len(x) - 1): continue if np.sign(diff) == 1: @@ -1029,24 +1046,25 @@ def plot_power_curves( data: dict[str, pd.DataFrame], power_col: str, windspeed_col: str, - flag_col: str = None, + flag_col: str | None = None, turbines: list[str] | None = None, flag_labels: tuple[str, str] = ("Flagged Readings", "Power Curve"), max_cols: int = 3, xlim: tuple[float, float] = (None, None), ylim: tuple[float, float] = (None, None), - legend: bool = False, - return_fig: bool = False, figure_kwargs: dict | None = None, legend_kwargs: dict | None = None, plot_kwargs: dict | None = None, + *, + legend: bool = False, + return_fig: bool = False, ): """Plots a series of power curves for a dictionary of turbine data, allowing for an optional filtering for singling out readings in the figure. Args: data(:obj:`dict[str, pd.DataFrame]`): The dictionary of turbine IDs and and SCADA data. - wind_speed_col(:obj:`pandas.Series`): A pandas Series or numpy array of the recorded wind + windspeed_col(:obj:`pandas.Series`): A pandas Series or numpy array of the recorded wind speeds, in m/s. power_col(:obj:`pandas.Series` | :obj:`np.ndarray`): A pandas Series or numpy array of the recorded power, in kW. @@ -1076,6 +1094,7 @@ def plot_power_curves( Returns: None | tuple[plt.Figure, plt.Axes]: Returns the figure and axes objects if :py:attr:`return_fig` is True. + """ if figure_kwargs is None: figure_kwargs = {} @@ -1092,7 +1111,7 @@ def plot_power_curves( figure_kwargs.setdefault("figsize", (15, num_rows * 5)) fig, axes_list = plt.subplots(num_rows, max_cols, **figure_kwargs) - for i, (t, ax) in enumerate(zip(turbines, axes_list.flatten())): + for i, (t, ax) in enumerate(zip(turbines, axes_list.flatten(), strict=False)): plot_data = data[t] label = "Power Curve" if flag_labels is None else flag_labels[1] @@ -1133,18 +1152,19 @@ def plot_wake_losses( bins: NDArrayFloat, efficiency_data_por: NDArrayFloat, efficiency_data_lt: NDArrayFloat, - energy_data_por: NDArrayFloat = None, - energy_data_lt: NDArrayFloat = None, + energy_data_por: NDArrayFloat | None = None, + energy_data_lt: NDArrayFloat | None = None, bin_axis_label: str = "wd", - turbine_id: str = None, + turbine_id: str | None = None, xlim: tuple[float, float] = (None, None), ylim_efficiency: tuple[float, float] = (None, None), ylim_energy: tuple[float, float] = (None, None), - return_fig: bool = False, figure_kwargs: dict | None = None, plot_kwargs_line: dict | None = None, plot_kwargs_fill: dict | None = None, legend_kwargs: dict | None = None, + *, + return_fig: bool = False, ): """Plots wake losses in the form of wind farm efficiency as well as normalized wind plant energy production for both the period of record and with the long-term correction as a function of either @@ -1201,6 +1221,7 @@ def plot_wake_losses( If :py:attr:`return_fig` is True, then the figure and axes object(s), corresponding to the wake loss plot or, if :py:attr:`energy_data_por` and :py:attr:`energy_data_lt` arguments are provided, wake loss and normalized energy plots, are returned for further tinkering/saving. + """ if figure_kwargs is None: figure_kwargs = {} @@ -1233,9 +1254,9 @@ def plot_wake_losses( # determine if normalized energy should be plotted if (energy_data_por is not None) & (energy_data_lt is not None): - if (not UQ) & (energy_data_por.ndim == 1) & (energy_data_lt.ndim == 1): - plot_norm_energy = True - elif UQ & (energy_data_por.ndim == 2) & (energy_data_lt.ndim == 2): + if (not UQ) & (energy_data_por.ndim == 1) & (energy_data_lt.ndim == 1) or UQ & ( + energy_data_por.ndim == 2 + ) & (energy_data_lt.ndim == 2): plot_norm_energy = True else: raise ValueError( @@ -1397,12 +1418,13 @@ def plot_yaw_misalignment( power_performance_label: str = "Normalized Cp (-)", xlim: tuple[float, float] = (None, None), ylim: tuple[float, float] = (None, None), - return_fig: bool = False, figure_kwargs: dict | None = None, plot_kwargs_curve: dict | None = None, plot_kwargs_line: dict | None = None, plot_kwargs_fill: dict | None = None, legend_kwargs: dict | None = None, + *, + return_fig: bool = False, ): """Plots power performance vs. wind vane angle along with the best-fit cosine curve for each wind speed bin for a single turbine. The mean wind vane angle and the wind vane angle where @@ -1459,8 +1481,8 @@ def plot_yaw_misalignment( None | tuple[matplotlib.pyplot.Figure, matplotlib.pyplot.Axes]: If `return_fig` is True, then the figure and axes object(s) corresponding to the yaw misalignment plots are returned for further tinkering/saving. - """ + """ from openoa.analysis.yaw_misalignment import cos_curve if figure_kwargs is None: @@ -1577,7 +1599,7 @@ def plot_yaw_misalignment( ], color=curve_fit_color_code, linestyle="--", - label=rf"Max. Power Vane Angle = {round(curve_fit_params_ws[:, i, 1].mean(), 1)}$^\circ$", # noqa: W605 + label=rf"Max. Power Vane Angle = {round(curve_fit_params_ws[:, i, 1].mean(), 1)}$^\circ$", ) yaw_mis_mean = np.round(np.mean(yaw_misalignment_ws[:, i]), 1) @@ -1586,7 +1608,7 @@ def plot_yaw_misalignment( ax.set_title( f"{ws} m/s\nYaw Misalignment = " - rf"{yaw_mis_mean}$^\circ$ [{yaw_mis_lb}$^\circ$, {yaw_mis_ub}$^\circ$]" # noqa: W605 + rf"{yaw_mis_mean}$^\circ$ [{yaw_mis_lb}$^\circ$, {yaw_mis_ub}$^\circ$]" ) else: norm_factor = curve_fit_params_ws[i, 0] @@ -1621,11 +1643,11 @@ def plot_yaw_misalignment( ], color=curve_fit_color_code, linestyle="--", - label=rf"Max. Power Vane Angle = {round(curve_fit_params_ws[i, 1], 1)}$^\circ$", # noqa: W605 + label=rf"Max. Power Vane Angle = {round(curve_fit_params_ws[i, 1], 1)}$^\circ$", ) ax.set_title( - f"{ws} m/s\nYaw Misalignment = {np.round(yaw_misalignment_ws[i], 1)}$^\\circ$" # noqa: W605 + f"{ws} m/s\nYaw Misalignment = {np.round(yaw_misalignment_ws[i], 1)}$^\\circ$" ) ax.plot( @@ -1636,7 +1658,7 @@ def plot_yaw_misalignment( ], color=mean_vane_color_code, linestyle="--", - label=rf"Mean Vane Angle = {round(mean_vane_angle_ws[i], 1)}$^\circ$", # noqa: W605 + label=rf"Mean Vane Angle = {round(mean_vane_angle_ws[i], 1)}$^\circ$", ) ax.grid("on") @@ -1662,13 +1684,13 @@ def plot_yaw_misalignment( for i in range(len(ws_bins) % 3, 3): axs[last_row][i].remove() axs[last_row - 1][i].tick_params(labelbottom=True) - axs[last_row - 1][i].set_xlabel(r"Wind Vane Angle ($^\circ$)") # noqa: W605 + axs[last_row - 1][i].set_xlabel(r"Wind Vane Angle ($^\circ$)") for i in range(len(ws_bins) % 3): - axs[last_row][i].set_xlabel(r"Wind Vane Angle ($^\circ$)") # noqa: W605 + axs[last_row][i].set_xlabel(r"Wind Vane Angle ($^\circ$)") else: for i in range(N_col): - axs[last_row][i].set_xlabel(r"Wind Vane Angle ($^\circ$)") # noqa: W605 + axs[last_row][i].set_xlabel(r"Wind Vane Angle ($^\circ$)") mean_yaw_mis = np.round(np.mean(yaw_misalignment_ws), 1) if UQ: @@ -1676,13 +1698,11 @@ def plot_yaw_misalignment( np.percentile(np.mean(yaw_misalignment_ws, 1), [2.5, 97.5]), 1 ) fig.suptitle( - rf"Turbine {turbine_id}, Yaw Misalignment = {mean_yaw_mis}$^\circ$ " # noqa: W605 - rf"[{yaw_misalignment_95CI[0]}$^\circ$, {yaw_misalignment_95CI[1]}$^\circ$]" # noqa: W605 + rf"Turbine {turbine_id}, Yaw Misalignment = {mean_yaw_mis}$^\circ$ " + rf"[{yaw_misalignment_95CI[0]}$^\circ$, {yaw_misalignment_95CI[1]}$^\circ$]" ) else: - fig.suptitle( - rf"Turbine {turbine_id}, Mean Yaw Misalignment = {str(mean_yaw_mis)}$^\circ$" # noqa: W605 - ) + fig.suptitle(rf"Turbine {turbine_id}, Mean Yaw Misalignment = {mean_yaw_mis!s}$^\circ$") plt.tight_layout() diff --git a/openoa/utils/power_curve/__init__.py b/openoa/utils/power_curve/__init__.py index d6cb49847..49f52b31f 100644 --- a/openoa/utils/power_curve/__init__.py +++ b/openoa/utils/power_curve/__init__.py @@ -1,5 +1,4 @@ -""" -This module provides methods to fit power curve models and use them to make predictions about 'ideal' +"""This module provides methods to fit power curve models and use them to make predictions about 'ideal' power generation. """ diff --git a/openoa/utils/power_curve/functions.py b/openoa/utils/power_curve/functions.py index 64a93748d..e9090f351 100644 --- a/openoa/utils/power_curve/functions.py +++ b/openoa/utils/power_curve/functions.py @@ -1,11 +1,10 @@ -""" -This module holds ready-to-use power curve functions. They take windspeed and power columns as arguments and return a +"""This module holds ready-to-use power curve functions. They take windspeed and power columns as arguments and return a python function which can be used to evaluate the power curve at arbitrary locations. """ from __future__ import annotations -from typing import Callable +from collections.abc import Callable import numpy as np import pandas as pd @@ -25,11 +24,11 @@ def IEC( bin_width: float = 0.5, windspeed_start: float = 0, windspeed_end: float = 30.0, - interpolate: bool = False, data: pd.DataFrame = None, + *, + interpolate: bool = False, ) -> Callable: - """ - Use IEC 61400-12-1-2 method for creating a binned wind-speed power curve. Power is set to zero + """Use IEC 61400-12-1-2 method for creating a binned wind-speed power curve. Power is set to zero for values outside the cutoff range: [:py:attr:`windspeed_start`, :py:attr:`windspeed_end`]. Args: @@ -50,7 +49,6 @@ def IEC( :obj:`Callable`: Python function of type (Array[float] -> Array[float]) implementing the power curve. """ - # Set up evenly spaced bins of fixed width, with any value over the maximum getting np.inf n_bins = int(np.ceil((windspeed_end - windspeed_start) / bin_width)) + 1 bins = np.append(np.linspace(windspeed_start, windspeed_end, n_bins), [np.inf]) @@ -59,7 +57,7 @@ def IEC( P_bin = np.ones(len(bins) - 1) * np.nan # Compute the mean of each bin and set corresponding P_bin - for ibin in range(0, len(bins) - 1): + for ibin in range(len(bins) - 1): indices = (windspeed_col >= bins[ibin]) & (windspeed_col < bins[ibin + 1]) P_bin[ibin] = power_col.loc[indices].mean() @@ -69,7 +67,7 @@ def IEC( # Create a closure over the computed bins which computes the power curve value for arbitrary array-like input def pc_iec_bin(x): P = np.zeros(np.shape(x)) - for i in range(0, len(bins) - 1): + for i in range(len(bins) - 1): idx = np.where((x >= bins[i]) & (x < bins[i + 1])) P[idx] = P_bin[i] cutoff_idx = (x < windspeed_start) | (x > windspeed_end) @@ -151,8 +149,7 @@ def gam( n_splines: int = 20, data: pd.DataFrame = None, ) -> Callable: - """ - Use the generalized additive model, :py:class:`pygam.LinearGAM` to fit power to wind speed. + """Use the generalized additive model, :py:class:`pygam.LinearGAM` to fit power to wind speed. Args: windspeed_col(:obj:`str` | `pandas.Series`): Windspeed data, or the name of the column in @@ -180,8 +177,7 @@ def gam_3param( n_splines: int = 20, data: pd.DataFrame = None, ) -> Callable: - """ - Use a generalized additive model to fit power to wind speed, wind direction and air density. + """Use a generalized additive model to fit power to wind speed, wind direction and air density. Args: windspeed_col(:obj:`str` | `pandas.Series`): Windspeed data, or the name of the column in @@ -199,6 +195,7 @@ def gam_3param( Returns: :obj:`Callable`: Python function of type (Array[float] -> Array[float]) implementing the power curve. + """ # create dataframe input to LinearGAM and predicted response variable X = data[[windspeed_col, wind_direction_col, air_density_col]] diff --git a/openoa/utils/power_curve/parametric_forms.py b/openoa/utils/power_curve/parametric_forms.py index 41b3c142c..a6bf07552 100644 --- a/openoa/utils/power_curve/parametric_forms.py +++ b/openoa/utils/power_curve/parametric_forms.py @@ -1,5 +1,4 @@ -""" -Power Curves +"""Power Curves. These curve functions are written in the style of Scipy.optimize: @@ -32,6 +31,7 @@ def _power_curve(x: np.ndarray | pd.Series, a: float, b: float, c: float, d: flo Returns: (:obj:`numpy.ndarray`): The converted power data. + """ if isinstance(x, pd.Series): x = x.values @@ -39,7 +39,7 @@ def _power_curve(x: np.ndarray | pd.Series, a: float, b: float, c: float, d: flo def logistic5param(x: np.ndarray | pd.Series, a: float, b: float, c: float, d: float, g: float): - """Create and return a 5 parameter logistic function + """Create and return a 5 parameter logistic function. Args: x(:obj:`numpy.ndarray` | `pandas.Series`): Input data. @@ -51,9 +51,7 @@ def logistic5param(x: np.ndarray | pd.Series, a: float, b: float, c: float, d: f Returns: Function[numpy.ndarray[real]] -> numpy.ndarray[real] - """ - res = np.ones_like(x, dtype=np.float64) # In the case where b<0, x==0, there is a divide by zero error. The answer should be "d" when x==0 and b<0. if b < 0: diff --git a/openoa/utils/power_curve/parametric_optimize.py b/openoa/utils/power_curve/parametric_optimize.py index dac6051d7..750c02a59 100644 --- a/openoa/utils/power_curve/parametric_optimize.py +++ b/openoa/utils/power_curve/parametric_optimize.py @@ -1,14 +1,13 @@ -""" -Curve fitting routines +"""Curve fitting routines. -curve + bounds -optimization algorithm -cost function +- curve + bounds +- optimization algorithm +- cost function """ from __future__ import annotations -from typing import Callable +from collections.abc import Callable import numpy as np import pandas as pd @@ -27,16 +26,16 @@ def fit_parametric_power_curve( tuple[float, float], tuple[float, float], ], + *, return_params: bool = False, ): - """ - Fit curve to filtered power-windspeed data. + """Fit curve to filtered power-wind speed data. Args: x(:obj:`numpy.ndarray` | `pandas.Series`): independent variable y(:obj:`numpy.ndarray` | `pandas.Series`): dependent variable curve(:obj:`Callable`): function/lambda name for power curve desired default is curves.logistic5param - optimization_algorithm(Function): scipy.optimize style optimization algorithm + optimization_algorithm(Callable): scipy.optimize style optimization algorithm cost_function(:obj:`Callable`): Python function that takes two np.array 1D of real numbers and returns a real numeric cost. bounds(:obj:`tuple[tuple[float, float], tuple[float, float], tuple[float, float], tuple[float, float], tuple[float, float]]`): @@ -46,6 +45,7 @@ def fit_parametric_power_curve( Returns: Callable(np.array -> np.array): function handle to optimized power curve + """ # Build opt function as a closure on "x" and "y" @@ -72,7 +72,7 @@ def fit_curve(x_2): def least_squares(x: np.ndarray | pd.Series, y: np.ndarray | pd.Series): - """Least Squares loss function + """Least Squares loss function. Args: x(:obj:`np.ndarray` | `pandas.Series`): 1-D array of numbers representing x @@ -80,5 +80,6 @@ def least_squares(x: np.ndarray | pd.Series, y: np.ndarray | pd.Series): Returns: The least square of x and y. + """ return np.sum((x - y) ** 2) diff --git a/openoa/utils/qa.py b/openoa/utils/qa.py index 594d86180..5635b611d 100644 --- a/openoa/utils/qa.py +++ b/openoa/utils/qa.py @@ -2,7 +2,7 @@ from __future__ import annotations -from typing import Tuple, Union +import contextlib from datetime import datetime import pytz @@ -15,10 +15,11 @@ from dateutil import tz from openoa.utils import timeseries as ts -from openoa.logging import logging, logged_method_call +from openoa.logging import logging from openoa.utils.plot import set_styling -Number = Union[int, float] + +Number = int | float logger = logging.getLogger(__name__) set_styling() @@ -38,6 +39,7 @@ def _remove_tz(df: pd.DataFrame, t_local_column: str) -> tuple[np.ndarray, np.nd Returns: :obj:`numpy.ndarray`: Truth array that can be used to filter the timestamps and subsequent values. :obj:`numpy.ndarray`: Array of timezone-naive python `datetime` objects. + """ arr = np.array( [ @@ -70,6 +72,7 @@ def _get_time_window(df, ix, hour_window, time_col, local_time_col, utc_time_col Returns: (:obj:`pandas.DataFrame`): The filtered DataFrame object + """ if ix.tz is None: col = time_col @@ -93,6 +96,7 @@ def determine_offset_dst(df: pd.DataFrame, local_tz: str) -> pd.DataFrames: Returns: (:obj:`pd.DataFrame`): The updated dataframe with "utc_offset" and "is_dst" columns created. + """ # The new column names _offset = "utc_offset" @@ -114,7 +118,7 @@ def determine_offset_dst(df: pd.DataFrame, local_tz: str) -> pd.DataFrames: def convert_datetime_column( - df: pd.DataFrame, time_col: str, local_tz: str, tz_aware: bool + df: pd.DataFrame, time_col: str, local_tz: str, *, tz_aware: bool ) -> pd.DataFrame: """Converts the passed timestamp data to a pandas-encoded Datetime, and creates a corresponding localized and UTC timestamp using the :py:attr:`time_field` column name with either @@ -137,6 +141,7 @@ def convert_datetime_column( - :py:attr:`time_col`_localized: The fully converted and localized timestamp column - utc_offset: The difference, in hours between the localized and UTC time - is_dst: Indicator for whether of not the timestamp is considered to be DST (``True``) or not (``False``) + """ # Create the necessary columns for processing t_utc = f"{time_col}_utc" @@ -178,7 +183,7 @@ def convert_datetime_column( def duplicate_time_identification( df: pd.DataFrame, time_col: str, id_col: str -) -> tuple[pd.Series, None | pd.Series, None | pd.Series]: +) -> tuple[pd.Series, pd.Series | None, pd.Series | None]: """Identifies the time duplications on the modified SCADA data frame to highlight the duplications from the original time data (:py:attr:`time_col`), the UTC timestamps, and the localized timestamps, if the latter are available. @@ -195,6 +200,7 @@ def duplicate_time_identification( timestamps based on the original timestamp column, the localized timestamp column (``None`` if the column does not exist), and the UTC-converted timestamp column (``None`` if the column does not exist). + """ # Create the necessary columns for processing t_utc = f"{time_col}_utc" @@ -215,7 +221,7 @@ def duplicate_time_identification( def gap_time_identification( df: pd.DataFrame, time_col: str, freq: str -) -> tuple[pd.Series, None | pd.Series, None | pd.Series]: +) -> tuple[pd.Series, pd.Series | None, pd.Series | None]: """Identifies the time gaps on the modified SCADA data frame to highlight the missing timestamps from the original time data (`time_col`), the UTC timestamps, and the localized timestamps, if the latter are available. @@ -232,6 +238,7 @@ def gap_time_identification( timestamps based on the original timestamp column, the localized timestamp column (``None`` if the column does not exist), and the UTC-converted timestamp column (``None`` if the column does not exist). + """ # Create the necessary columns for processing t_utc = f"{time_col}_utc" @@ -260,6 +267,7 @@ def describe(df: pd.DataFrame, **kwargs) -> pd.DataFrame: Returns: pd.DataFrame: The results of ``df.describe().T``. + """ return df.describe(**kwargs).T @@ -290,6 +298,7 @@ def daylight_savings_plot( the pandas timestamp conventions (https://pandas.pydata.org/pandas-docs/stable/user_guide/timeseries.html#timeseries-offset-aliases). hour_window(:obj: 'int'): number of hours, before and after the Daylight Savings Time transitions to view in the plot, by default 3. + """ # Create the necessary columns for processing _dst = "is_dst" @@ -299,10 +308,8 @@ def daylight_savings_plot( # Get data for one of the turbines df_dst = df.loc[df[id_col] == df[id_col].unique()[0]] df_full = df_dst.copy() - try: + with contextlib.suppress(TypeError): df_full = df_full.tz_convert(local_tz) - except TypeError: - pass time_duplications, time_duplications_utc, _ = duplicate_time_identification( df, time_col, id_col @@ -313,7 +320,7 @@ def daylight_savings_plot( j = 0 fig = plt.figure(figsize=(20, 24)) - axes = axes = fig.subplots(num_years, 2, gridspec_kw=dict(wspace=0.15, hspace=0.3)) + axes = axes = fig.subplots(num_years, 2, gridspec_kw={"wspace": 0.15, "hspace": 0.3}) for i, year in enumerate(years): year_data = df_full.loc[df_full[time_col].dt.year == year] dst_dates = np.where(year_data[_dst].values)[0] @@ -482,6 +489,7 @@ def wtk_coordinate_indices( Returns: tuple[float, float]: The nearest valid x and y coordinates to the provided `latitude` and `longitude`. + """ coordinates = fn["coordinates"] project_coord_string = """ @@ -495,7 +503,7 @@ def wtk_coordinate_indices( project_coords = projectLcc(longitude, latitude) delta = np.subtract(project_coords, origin) - xy = reversed([int(round(x / 2000)) for x in delta]) + xy = reversed([round(x / 2000) for x in delta]) return tuple(xy) @@ -521,6 +529,7 @@ def wtk_diurnal_prep( Returns: pd.Series: The diurnal hourly average wind speed. + """ # Startup the API and grab the database f = h5pyd.File(fn, "r") @@ -535,9 +544,9 @@ def wtk_diurnal_prep( project_ix = wtk_coordinate_indices(f, latitude, longitude) try: _ = wtk_coordinates[project_ix[0]][project_ix[1]] - except ValueError: + except ValueError as e: msg = f"Project Coordinates (lat, long) = ({latitude}, {longitude}) are outside the WIND Toolkit domain." - raise IndexError(msg) + raise IndexError(msg) from e window_ix = dt.loc[(dt.datetime >= start_date) & (dt.datetime <= end_date)].index ws = pd.DataFrame( @@ -562,7 +571,7 @@ def wtk_diurnal_plot( end_date: str = "2013-12-31", return_fig: bool = False, ) -> None: - """Plots the WTK diurnal wind profile alongside the hourly power averages from the :py:attr:`scada_df` + """Plots the WTK diurnal wind profile alongside the hourly power averages from the :py:attr:`scada_df`. Args: wtk_df (:obj: `pd.DataFrame` | `None`): The WTK diurnal profile data produced in @@ -581,6 +590,7 @@ def wtk_diurnal_plot( uses the ending date of :py:attr:`scada_df`. Defaults to None. return_fig(:obj:`String`): Indicator for if the figure and axes objects should be returned, by default False. + """ # Get the WTK data if needed if wtk_df is None: diff --git a/openoa/utils/timeseries.py b/openoa/utils/timeseries.py index e1a7501c4..4b74b00ed 100644 --- a/openoa/utils/timeseries.py +++ b/openoa/utils/timeseries.py @@ -1,6 +1,4 @@ -""" -This module provides useful functions for processing timeseries data -""" +"""This module provides useful functions for processing timeseries data.""" from __future__ import annotations @@ -24,6 +22,7 @@ def offset_to_seconds(offset: int | float | str | np.datetime64) -> int | float: Returns: :obj:`int` | `float`: The number of seconds corresponding to :py:attr:`offset`. + """ try: seconds = pd.to_timedelta(offset).total_seconds() @@ -43,6 +42,7 @@ def determine_frequency_seconds(data: pd.DataFrame, index_col: str | None = None Returns: :obj:`int` | `float`: The number of seconds corresponding to :py:attr:`offset`. + """ # Get the non-duplicated DatetimeIndex values from a single level, or multi-level index index = data.index if index_col is None else data.index.get_level_values(index_col) @@ -63,6 +63,7 @@ def determine_frequency(data: pd.DataFrame, index_col: str | None = None) -> str Returns: :obj:`str` | :obj:`int` | :obj:`float`: The offset string or number of seconds between timestamps. + """ # Get the timetamp index values index = data.index if index_col is None else data.index.get_level_values(index_col) @@ -80,16 +81,17 @@ def determine_frequency(data: pd.DataFrame, index_col: str | None = None) -> str def convert_local_to_utc(d: str | datetime.datetime, tz_string: str) -> datetime.datetime: - """ - Convert timestamps in local time to UTC. The function can only act on a single timestamp at a time, so + """Convert timestamps in local time to UTC. + + The function can only act on a single timestamp at a time, so for example use the .apply function in Pandas: - date_utc = df['time'].apply(convert_local_to_utc, args = ('US/Pacific',)) + ``date_utc = df['time'].apply(convert_local_to_utc, args = ('US/Pacific',))`` Also note that this function doesn't solve the end of DST when times between 1:00-2:00 are repeated in November. Those dates are left repeated in UTC time and need to be shifted manually. - The function does address the missing 2:00-3:00 times at the start of DST in March + The function does address the missing 2:00-3:00 times at the start of DST in March. Args: d(:obj:`datetime.datetime`): the local date, tzinfo must not be set @@ -130,6 +132,7 @@ def convert_dt_to_utc( Returns: pd.Series: _description_ + """ if isinstance(dt_col[0], str): dt_col = dt_col.apply(parse) @@ -143,8 +146,7 @@ def convert_dt_to_utc( @series_method(data_cols=["dt_col"]) def find_time_gaps(dt_col: pd.Series | str, freq: str, data: pd.DataFrame = None) -> pd.Series: - """ - Finds gaps in `dt_col` based on the expected frequency, `freq`, and returns them. + """Finds gaps in `dt_col` based on the expected frequency, `freq`, and returns them. Args: dt_col(:obj:`pandas.Series`): Pandas ``Series`` of ``datetime.datetime`` objects or the name @@ -156,6 +158,7 @@ def find_time_gaps(dt_col: pd.Series | str, freq: str, data: pd.DataFrame = None Returns: :obj:`pandas.Series`: Series of missing time stamps in ``datetime.datetime`` format + """ if isinstance(dt_col, pd.DatetimeIndex): dt_col = dt_col.to_series() @@ -172,8 +175,7 @@ def find_time_gaps(dt_col: pd.Series | str, freq: str, data: pd.DataFrame = None @series_method(data_cols=["dt_col"]) def find_duplicate_times(dt_col: pd.Series | str, data: pd.DataFrame = None): - """ - Find duplicate input data and report them. The first duplicated item is not reported, only subsequent duplicates. + """Find duplicate input data and report them. The first duplicated item is not reported, only subsequent duplicates. Args: dt_col(:obj:`pandas.Series` | `str`): Pandas series of ``datetime.datetime`` objects or the name of the @@ -183,6 +185,7 @@ def find_duplicate_times(dt_col: pd.Series | str, data: pd.DataFrame = None): Returns: :obj:`pandas.Series`: Duplicates from input data + """ if isinstance(dt_col, pd.DatetimeIndex): dt_col = dt_col.to_series() @@ -191,8 +194,7 @@ def find_duplicate_times(dt_col: pd.Series | str, data: pd.DataFrame = None): def gap_fill_data_frame(data: pd.DataFrame, dt_col: str, freq: str) -> pd.DataFrame: - """ - Insert any missing timestamps into :py:attr:`data` while filling the data columns with NaNs. + """Insert any missing timestamps into :py:attr:`data` while filling the data columns with NaNs. Args: data(:obj:`pandas.DataFrame`): The dataframe with potentially missing timestamps. @@ -227,8 +229,7 @@ def gap_fill_data_frame(data: pd.DataFrame, dt_col: str, freq: str) -> pd.DataFr @series_method(data_cols=["col"]) def percent_nan(col: pd.Series | str, data: pd.DataFrame = None): - """ - Return percentage of data that are Nan or 1 if the series is empty. + """Return percentage of data that are Nan or 1 if the series is empty. Args: col(:obj:`pandas.Series`): The pandas `Series` to be checked for NaNs, or the name of the @@ -238,14 +239,14 @@ def percent_nan(col: pd.Series | str, data: pd.DataFrame = None): Returns: :obj:`float`: Percentage of NaN data in the data series + """ return 1 if (denominator := float(col.size)) == 0 else np.isnan(col.values).sum() / denominator @series_method(data_cols=["dt_col"]) def num_days(dt_col: pd.Series | str, data: pd.DataFrame = None) -> int: - """ - Calculates the number of non-duplicate days in :py:attr:`dt_col`. + """Calculates the number of non-duplicate days in :py:attr:`dt_col`. Args: dt_col(:obj:`pandas.Series` | str): A pandas ``Series`` with a timeseries index to be checked @@ -255,14 +256,14 @@ def num_days(dt_col: pd.Series | str, data: pd.DataFrame = None) -> int: Returns: :obj:`int`: Number of days in the data + """ return dt_col[~dt_col.index.duplicated()].resample("D").asfreq().index.size @series_method(data_cols=["dt_col"]) def num_hours(dt_col: pd.Series | str, *, data: pd.DataFrame = None) -> int: - """ - Calculates the number of non-duplicate hours in `dt_col`. + """Calculates the number of non-duplicate hours in `dt_col`. Args: dt_col(:obj:`pandas.Series` | str): A pandas ``Series`` of timeseries data to be checked for @@ -272,5 +273,6 @@ def num_hours(dt_col: pd.Series | str, *, data: pd.DataFrame = None) -> int: Returns: :obj:`int`: Number of hours in the data + """ return dt_col[~dt_col.index.duplicated()].resample("h").asfreq().index.size diff --git a/openoa/utils/unit_conversion.py b/openoa/utils/unit_conversion.py index 7b1fccd80..0234f70e0 100644 --- a/openoa/utils/unit_conversion.py +++ b/openoa/utils/unit_conversion.py @@ -1,5 +1,5 @@ -""" -This module provides basic methods for unit conversion and calculation of basic wind plant variables +"""This module provides basic methods for unit conversion and calculation of basic wind plant +variables. """ from __future__ import annotations @@ -14,8 +14,7 @@ def convert_power_to_energy( power_col: str | pd.Series, sample_rate_min="10min", data: pd.DataFrame = None ) -> pd.Series: - """ - Compute energy [kWh] from power [kw] and return the data column + """Compute energy [kWh] from power [kw] and return the data column. Args: power_col(:obj:`str` | :obj:`pandas.Series`): The power data, in kW, or the name of the column @@ -46,8 +45,7 @@ def compute_gross_energy( curtailment_type: str = "frac", data: str | pd.DataFrame = None, ): - """ - Computes gross energy for a wind plant or turbine by adding reported :py:attr:`availability` and + """Computes gross energy for a wind plant or turbine by adding reported :py:attr:`availability` and :py:attr:`curtailment` losses to reported net energy. Args: @@ -68,6 +66,7 @@ def compute_gross_energy( Returns: gross(:obj:`pandas.Series`): Calculated gross energy for wind plant or turbine + """ if np.any(availability < 0) | np.any(curtailment < 0): raise ValueError( @@ -91,8 +90,7 @@ def compute_gross_energy( @series_method(data_cols=["variable"]) def convert_feet_to_meter(variable: str | pd.Series, data: pd.DataFrame = None): - """ - Compute variable in [meter] from [feet] and return the data column + """Compute variable in [meter] from [feet] and return the data column. Args: variable(:obj:`str` | `pandas.Series`): A pandas Series, the name of the columnn in @@ -102,5 +100,6 @@ def convert_feet_to_meter(variable: str | pd.Series, data: pd.DataFrame = None): Returns: :obj:`pandas.Series`: :py:attr:`variable` in meters + """ return variable * 0.3048 diff --git a/pyproject.toml b/pyproject.toml index c80768ede..306e08368 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -70,10 +70,8 @@ changelog = "https://github.com/NREL/OpenOA/blob/main/CHANGELOG.md" [project.optional-dependencies] develop = [ "pre-commit", - "black", + "ruff", "isort", - "flake8", - "flake8-docstrings", "pytest>=9", "pytest-cov>=2.8.1", ] @@ -139,40 +137,18 @@ filterwarnings = [ "ignore::DeprecationWarning:pygam.*:", ] -[tool.black] -# https://github.com/psf/black -line-length = 100 -target-version = ["py310", "py311", "py312", "py313"] -include = '\.pyi?$' -exclude = ''' -# A regex preceded with ^/ will apply only to files and directories -# in the root of the project. -^/( - ( - \.eggs # exclude a few common directories in the - | \.git # root of the project - | \.hg - | \.mypy_cache - | \.tox - | \.venv - | _build - | buck-out - | build - | dist - )/ - | foo.py # also separately exclude a file named foo.py in - # the root of the project -) -''' - [tool.isort] # https://github.com/PyCQA/isort -profile = "black" py_version = 310 src_paths = ["isort", "test"] -line_length = "100" -length_sort = "True" -length_sort_straight = "True" +combine_as_imports = true +multi_line_output = 3 +include_trailing_comma = true +use_parentheses = true +length_sort = true +length_sort_straight = true +lines_after_imports = 2 +line_length = 100 sections = ["FUTURE", "STDLIB", "THIRDPARTY", "FIRSTPARTY", "LOCALFOLDER"] known_first_party = "openoa" known_third_party = [ @@ -188,3 +164,74 @@ known_third_party = [ "pyproj", "shapely" ] + + +[tool.ruff] +src = ["openoa", "test"] +line-length = 100 +target-version = "py310" +fix = true + +[tool.ruff.lint] +select = [ + "F", + "E", + "W", + "C4", + "D", + "UP", + "NPY", + "PTH", + "RUF", + "FBT", + "B", + "PIE", + "T20", + "SIM", +] + +# D205: not using summary lines and descriptions, just descriptions +# D401: don't believe enough in imperative mode to make all the changes currently +# E501: > 550 line length errors from comments and docs, only use to fix lines, then turn off +# NPY002: Need to adopt NumPy >= 2 first +ignore = ["D205", "D401", "NPY002", "E501"] + +# Allow autofix for all enabled rules (when `--fix`) is provided. +fixable = ["F", "E", "W", "C4", "D", "UP", "NPY", "PTH", "RUF", "FBT", "B", "PIE", "T20", "SIM", ] +unfixable = [] + +# Exclude a variety of commonly ignored directories. +exclude = [ + ".bzr", + ".direnv", + ".eggs", + ".git", + ".hg", + ".mypy_cache", + ".nox", + ".pants.d", + ".ruff_cache", + ".svn", + ".tox", + ".venv", + "__pypackages__", + "_build", + "buck-out", + "build", + "dist", + "node_modules", + "venv", + "sphinx/examples/*.ipynb", + "examples/*.ipynb" +] + +# Allow unused variables when underscore-prefixed. +dummy-variable-rgx = "^(_+|(_+[a-zA-Z0-9_]*[a-zA-Z0-9]+?))$" + +[tool.ruff.lint.per-file-ignores] +"*/__init__.py" = ["E501", "D104", "F401", "D415"] +"test/*" = ["D100", "D101", "D102", "D103"] +"examples/*.ipynb" = ["T201"] + +[tool.ruff.lint.pydocstyle] +convention = "google" diff --git a/sphinx/api/utils.rst b/sphinx/api/utils.rst index 07f1c0031..621a3cdc2 100644 --- a/sphinx/api/utils.rst +++ b/sphinx/api/utils.rst @@ -8,7 +8,7 @@ The utils subpackage provides module-level methods that operate on Pandas `DataF be imported and used individually into your own scripts. Downloader -******* +********** .. automodule:: openoa.utils.downloader :members: diff --git a/sphinx/conf.py b/sphinx/conf.py index e1d6c8a8d..e2c8de881 100644 --- a/sphinx/conf.py +++ b/sphinx/conf.py @@ -1,27 +1,18 @@ -# -# Operational Analysis documentation build configuration file, created by -# sphinx-quickstart on Fri Dec 8 11:30:30 2017. -# -# This file is execfile()d with the current directory set to its -# containing dir. -# -# Note that not all possible configuration values are present in this -# autogenerated file. -# -# All configuration values have a default; values that are commented out -# serve to show the default. +"""Sphinx configuration file. +Operational Analysis documentation build configuration file, created by sphinx-quickstart on Fri Dec 8 11:30:30 2017. +This file is execfile()d with the current directory set to its containing dir. +Note that not all possible configuration values are present in this autogenerated file. +All configuration values have a default; values that are commented out serve to show the default. +If extensions (or modules to document with autodoc) are in another directory, add these directories to sys.path here. If the directory is relative to the documentation root, use os.path.abspath to make it absolute, like shown here. +""" -# If extensions (or modules to document with autodoc) are in another directory, -# add these directories to sys.path here. If the directory is relative to the -# documentation root, use os.path.abspath to make it absolute, like shown here. -# -import io import os import re import sys from pathlib import Path -sys.path.insert(0, os.path.abspath("..")) + +sys.path.insert(0, os.path.abspath("..")) # noqa: PTH100 # -- General configuration ------------------------------------------------ @@ -70,17 +61,21 @@ # The version info for the project you're documenting, acts as replacement for # |version| and |release|, also used in various other places throughout the # built documents. -# -# Read the version from the __init__.py file without importing it + + def read(*names, **kwargs): - with open( - os.path.join(os.path.dirname(__file__)[:-7], *names), - encoding=kwargs.get("encoding", "utf8"), - ) as fp: + """Read the version from the __init__.py file without importing it.""" + with ( + open( # noqa: PTH123 + os.path.join(os.path.dirname(__file__)[:-7], *names), # noqa: PTH118, PTH120 + encoding=kwargs.get("encoding", "utf8"), + ) as fp + ): return fp.read() def find_version(*file_paths): + """Retrieve the version from a series of files.""" version_file = read(*file_paths) version_match = re.search(r"^__version__ = ['\"]([^'\"]*)['\"]", version_file, re.M) if version_match: diff --git a/sphinx/examples/00_intro_to_plant_data.ipynb b/sphinx/examples/00_intro_to_plant_data.ipynb index 11758d8f7..69fddf351 100644 --- a/sphinx/examples/00_intro_to_plant_data.ipynb +++ b/sphinx/examples/00_intro_to_plant_data.ipynb @@ -141,7 +141,6 @@ "source": [ "from pprint import pprint\n", "\n", - "import numpy as np\n", "import pandas as pd\n", "from matplotlib import pyplot as plt\n", "\n", @@ -809,7 +808,7 @@ " df=scada_df_tz,\n", " time_col=\"Date_time\",\n", " local_tz=\"Europe/Paris\",\n", - " tz_aware=True # Indicate that we can use encoded data to convert between timezones\n", + " tz_aware=True, # Indicate that we can use encoded data to convert between timezones\n", ")\n", "scada_df_tz.head()" ] @@ -1043,7 +1042,7 @@ " df=scada_df_no_tz,\n", " time_col=\"Date_time\",\n", " local_tz=\"Europe/Paris\",\n", - " tz_aware=False # Indicates that we're going to need to make inferences about encoding the timezones\n", + " tz_aware=False, # Indicates that we're going to need to make inferences about encoding the timezones\n", ")\n", "scada_df_no_tz.head()" ] @@ -1249,8 +1248,19 @@ ], "source": [ "no_tz = qa.describe(scada_df_no_tz)\n", - "no_tz = no_tz.loc[~no_tz.index.isin([\"Date_time\"])] # Ignore the Date_time column that is not shared between the dataframes\n", - "col_order = [\"count\", \"mean\", \"std\", \"min\", \"25%\", \"50%\", \"75%\", \"max\"] # Ensure description columns are in the same order\n", + "no_tz = no_tz.loc[\n", + " ~no_tz.index.isin([\"Date_time\"])\n", + "] # Ignore the Date_time column that is not shared between the dataframes\n", + "col_order = [\n", + " \"count\",\n", + " \"mean\",\n", + " \"std\",\n", + " \"min\",\n", + " \"25%\",\n", + " \"50%\",\n", + " \"75%\",\n", + " \"max\",\n", + "] # Ensure description columns are in the same order\n", "qa.describe(scada_df_tz)[col_order] == no_tz[col_order]" ] }, @@ -1621,9 +1631,7 @@ ], "source": [ "dup_orig_no_tz, dup_local_no_tz, dup_utc_no_tz = qa.duplicate_time_identification(\n", - " df=scada_df_no_tz,\n", - " time_col=\"Date_time\",\n", - " id_col=\"Wind_turbine_name\"\n", + " df=scada_df_no_tz, time_col=\"Date_time\", id_col=\"Wind_turbine_name\"\n", ")\n", "dup_orig_no_tz.size, dup_local_no_tz.size, dup_utc_no_tz.size" ] @@ -1729,9 +1737,7 @@ ], "source": [ "gap_orig_no_tz, gap_local_no_tz, gap_utc_no_tz = qa.gap_time_identification(\n", - " df=scada_df_no_tz,\n", - " time_col=\"Date_time\",\n", - " freq=\"10min\"\n", + " df=scada_df_no_tz, time_col=\"Date_time\", freq=\"10min\"\n", ")\n", "gap_orig_no_tz.size, gap_local_no_tz.size, gap_utc_no_tz.size" ] @@ -1840,7 +1846,7 @@ " time_col=\"Date_time\",\n", " power_col=\"P_avg\",\n", " freq=\"10min\",\n", - " hour_window=3 # default value\n", + " hour_window=3, # default value\n", ")" ] }, @@ -1864,9 +1870,7 @@ "outputs": [], "source": [ "dup_orig_tz, dup_local_tz, dup_utc_tz = qa.duplicate_time_identification(\n", - " df=scada_df_tz,\n", - " time_col=\"Date_time\",\n", - " id_col=\"Wind_turbine_name\"\n", + " df=scada_df_tz, time_col=\"Date_time\", id_col=\"Wind_turbine_name\"\n", ")" ] }, @@ -1980,9 +1984,7 @@ ], "source": [ "gap_orig_tz, gap_local_tz, gap_utc_tz = qa.gap_time_identification(\n", - " df=scada_df_tz,\n", - " time_col=\"Date_time\",\n", - " freq=\"10min\"\n", + " df=scada_df_tz, time_col=\"Date_time\", freq=\"10min\"\n", ")\n", "gap_orig_tz.size, gap_local_tz.size, gap_utc_tz.size" ] @@ -2054,7 +2056,7 @@ " time_col=\"Date_time\",\n", " power_col=\"P_avg\",\n", " freq=\"10min\",\n", - " hour_window=3 # default value\n", + " hour_window=3, # default value\n", ")" ] }, @@ -2265,6 +2267,7 @@ ], "source": [ "from openoa.plant import PlantMetaData\n", + "\n", "print(PlantMetaData.__doc__)" ] }, @@ -2349,7 +2352,7 @@ "print(f\"The new analysis types now has all and MonteCarloAEP: {engie.analysis_type}\")\n", "try:\n", " engie.validate()\n", - "except ValueError as e: # Catch the error message so that the whole notebook can run\n", + "except ValueError as e: # Catch the error message so that the whole notebook can run\n", " print(e)" ] }, @@ -2398,7 +2401,7 @@ " asset=f\"{data_path}/asset.csv\",\n", " reanalysis={\n", " \"era5\": f\"{data_path}/reanalysis_era5.csv\",\n", - " \"merra2\": f\"{data_path}/reanalysis_merra2.csv\"\n", + " \"merra2\": f\"{data_path}/reanalysis_merra2.csv\",\n", " },\n", ")" ] diff --git a/sphinx/examples/01_utils_examples.ipynb b/sphinx/examples/01_utils_examples.ipynb index f3c7e26ae..8ccab56a3 100644 --- a/sphinx/examples/01_utils_examples.ipynb +++ b/sphinx/examples/01_utils_examples.ipynb @@ -363,10 +363,10 @@ "source": [ "import matplotlib.pyplot as plt\n", "import numpy as np\n", - "import pandas as pd\n", "\n", "from bokeh.plotting import show\n", "from bokeh.io import output_notebook\n", + "\n", "output_notebook()\n", "\n", "from openoa.utils import filters, power_curve, plot\n", @@ -697,7 +697,7 @@ " xlim=(-1, 20), # optional input for refining plots\n", " ylim=(-100, 2100), # optional input for refining plots\n", " legend=True, # optional flag for adding a legend\n", - " scatter_kwargs=dict(alpha=0.8, s=10) # optional input for refining plots\n", + " scatter_kwargs=dict(alpha=0.8, s=10), # optional input for refining plots\n", ")" ] }, @@ -757,7 +757,7 @@ } ], "source": [ - "out_of_window = filters.window_range_flag(windspeed, 5., 40, power_kw, 20., 2100.)\n", + "out_of_window = filters.window_range_flag(windspeed, 5.0, 40, power_kw, 20.0, 2100.0)\n", "plot.plot_power_curve(\n", " windspeed,\n", " power_kw,\n", @@ -766,7 +766,7 @@ " xlim=(-1, 20), # optional input for refining plots\n", " ylim=(-100, 2100), # optional input for refining plots\n", " legend=True, # optional flag for adding a legend\n", - " scatter_kwargs=dict(alpha=0.4, s=10) # optional input for refining plots\n", + " scatter_kwargs=dict(alpha=0.4, s=10), # optional input for refining plots\n", ")" ] }, @@ -816,7 +816,9 @@ ], "source": [ "max_bin = 0.90 * power_kw_filt1.max()\n", - "bin_outliers = filters.bin_filter(power_kw_filt1, windspeed_filt1, 100, 1.5, \"median\", 20., max_bin, \"scalar\", \"all\")\n", + "bin_outliers = filters.bin_filter(\n", + " power_kw_filt1, windspeed_filt1, 100, 1.5, \"median\", 20.0, max_bin, \"scalar\", \"all\"\n", + ")\n", "plot.plot_power_curve(\n", " windspeed_filt1,\n", " power_kw_filt1,\n", @@ -825,7 +827,7 @@ " xlim=(-1, 20), # optional input for refining plots\n", " ylim=(-100, 2100), # optional input for refining plots\n", " legend=True, # optional flag for adding a legend\n", - " scatter_kwargs=dict(alpha=0.5, s=10) # optional input for refining plots\n", + " scatter_kwargs=dict(alpha=0.5, s=10), # optional input for refining plots\n", ")" ] }, @@ -881,7 +883,7 @@ " xlim=(-1, 20), # optional input for refining plots\n", " ylim=(-100, 2100), # optional input for refining plots\n", " legend=True, # optional flag for adding a legend\n", - " scatter_kwargs=dict(alpha=0.4, s=10) # optional input for refining plots\n", + " scatter_kwargs=dict(alpha=0.4, s=10), # optional input for refining plots\n", ")" ] }, @@ -972,9 +974,9 @@ ")\n", "\n", "x = np.linspace(0, 20, 100)\n", - "ax.plot(x, iec_curve(x), color=\"red\", label = \"IEC\", linewidth = 3)\n", - "ax.plot(x, spline_curve(x), color=\"C1\", label = \"Spline\", linewidth = 3)\n", - "ax.plot(x, l5p_curve(x), color=\"C2\", label = \"L5P\", linewidth = 3)\n", + "ax.plot(x, iec_curve(x), color=\"red\", label=\"IEC\", linewidth=3)\n", + "ax.plot(x, spline_curve(x), color=\"C1\", label=\"Spline\", linewidth=3)\n", + "ax.plot(x, l5p_curve(x), color=\"C2\", label=\"L5P\", linewidth=3)\n", "\n", "ax.legend()\n", "\n", diff --git a/sphinx/examples/02a_plant_aep_analysis.ipynb b/sphinx/examples/02a_plant_aep_analysis.ipynb index b388fae6b..9029514dc 100644 --- a/sphinx/examples/02a_plant_aep_analysis.ipynb +++ b/sphinx/examples/02a_plant_aep_analysis.ipynb @@ -41,15 +41,11 @@ "metadata": {}, "outputs": [], "source": [ - "import os\n", - "import copy\n", "from datetime import datetime\n", "\n", "import numpy as np\n", "import pandas as pd\n", "import matplotlib.pyplot as plt\n", - "import statsmodels.api as sm\n", - "from IPython.display import clear_output\n", "\n", "from openoa.analysis.aep import MonteCarloAEP\n", "from openoa.utils import plot\n", @@ -71,7 +67,7 @@ "outputs": [], "source": [ "# Load plant object\n", - "project = project_ENGIE.prepare('./data/la_haute_borne', use_cleansed=False)" + "project = project_ENGIE.prepare(\"./data/la_haute_borne\", use_cleansed=False)" ] }, { @@ -213,7 +209,7 @@ "metadata": {}, "outputs": [], "source": [ - "pa = MonteCarloAEP(project, reanalysis_products = ['era5', 'merra2'])" + "pa = MonteCarloAEP(project, reanalysis_products=[\"era5\", \"merra2\"])" ] }, { @@ -520,7 +516,9 @@ } ], "source": [ - "pa.plot_reanalysis_gross_energy_data(outlier_threshold=3, xlim=(4, 9), ylim=(0, 2), plot_kwargs=dict(s=60))" + "pa.plot_reanalysis_gross_energy_data(\n", + " outlier_threshold=3, xlim=(4, 9), ylim=(0, 2), plot_kwargs=dict(s=60)\n", + ")" ] }, { @@ -550,9 +548,7 @@ ], "source": [ "pa.plot_aggregate_plant_data_timeseries(\n", - " xlim=(datetime(2013, 12, 1), datetime(2015, 12, 31)),\n", - " ylim_energy=(0, 2),\n", - " ylim_loss=(-0.1, 5.5)\n", + " xlim=(datetime(2013, 12, 1), datetime(2015, 12, 31)), ylim_energy=(0, 2), ylim_loss=(-0.1, 5.5)\n", ")" ] }, @@ -576,8 +572,8 @@ "outputs": [], "source": [ "# For illustrative purposes, let's suppose a few months aren't representative of long-term losses\n", - "pa.aggregate.loc['2014-11-01',['availability_typical','curtailment_typical']] = False\n", - "pa.aggregate.loc['2015-07-01',['availability_typical','curtailment_typical']] = False" + "pa.aggregate.loc[\"2014-11-01\", [\"availability_typical\", \"curtailment_typical\"]] = False\n", + "pa.aggregate.loc[\"2015-07-01\", [\"availability_typical\", \"curtailment_typical\"]] = False" ] }, { @@ -654,7 +650,7 @@ ], "source": [ "# Run Monte Carlo based OA\n", - "pa.run(num_sim=2000, reanalysis_products=['era5', 'merra2'])" + "pa.run(num_sim=2000, reanalysis_products=[\"era5\", \"merra2\"])" ] }, { @@ -729,16 +725,18 @@ "outputs": [], "source": [ "# Produce histograms of the various MC-parameters\n", - "mc_reg = pd.DataFrame(data={\n", - " 'slope': pa._mc_slope.ravel(),\n", - " 'intercept': pa._mc_intercept, \n", - " 'num_points': pa._mc_num_points, \n", - " 'metered_energy_fraction': pa.mc_inputs.metered_energy_fraction, \n", - " 'loss_fraction': pa.mc_inputs.loss_fraction,\n", - " 'num_years_windiness': pa.mc_inputs.num_years_windiness, \n", - " 'loss_threshold': pa.mc_inputs.loss_threshold,\n", - " 'reanalysis_product': pa.mc_inputs.reanalysis_product\n", - "})" + "mc_reg = pd.DataFrame(\n", + " data={\n", + " \"slope\": pa._mc_slope.ravel(),\n", + " \"intercept\": pa._mc_intercept,\n", + " \"num_points\": pa._mc_num_points,\n", + " \"metered_energy_fraction\": pa.mc_inputs.metered_energy_fraction,\n", + " \"loss_fraction\": pa.mc_inputs.loss_fraction,\n", + " \"num_years_windiness\": pa.mc_inputs.num_years_windiness,\n", + " \"loss_threshold\": pa.mc_inputs.loss_threshold,\n", + " \"reanalysis_product\": pa.mc_inputs.reanalysis_product,\n", + " }\n", + ")" ] }, { @@ -800,17 +798,18 @@ "# Produce scatter plots of slope and intercept values. Here we focus on the ERA-5 data\n", "plot.set_styling()\n", "\n", - "plt.figure(figsize=(8,6))\n", + "plt.figure(figsize=(8, 6))\n", "plt.plot(\n", - " mc_reg.intercept[mc_reg.reanalysis_product =='era5'],\n", - " mc_reg.slope[mc_reg.reanalysis_product =='era5'],\n", - " '.', label=\"Monte Carlo Regression Values\"\n", + " mc_reg.intercept[mc_reg.reanalysis_product == \"era5\"],\n", + " mc_reg.slope[mc_reg.reanalysis_product == \"era5\"],\n", + " \".\",\n", + " label=\"Monte Carlo Regression Values\",\n", ")\n", "x = np.linspace(-2, 0, 3)\n", "y = -0.2 * x + 0.135\n", "plt.plot(x, y, label=\"y = -0.2x + 0.135\")\n", - "plt.xlabel('Intercept (GWh)')\n", - "plt.ylabel('Slope (GWh / (m/s))')\n", + "plt.xlabel(\"Intercept (GWh)\")\n", + "plt.ylabel(\"Slope (GWh / (m/s))\")\n", "plt.legend()\n", "plt.xlim((-1.8, -0.4))\n", "plt.ylim(0.2, 0.5)\n", @@ -841,7 +840,7 @@ } ], "source": [ - "pa.plot_aep_boxplot(x=mc_reg['reanalysis_product'], xlabel=\"Reanalysis Product\", ylim=(6, 18))" + "pa.plot_aep_boxplot(x=mc_reg[\"reanalysis_product\"], xlabel=\"Reanalysis Product\", ylim=(6, 18))" ] }, { @@ -874,11 +873,11 @@ "# NOTE: This is the same method, but calling the same method through the plot module directly\n", "plot.plot_boxplot(\n", " y=pa.results.aep_GWh,\n", - " x=mc_reg['num_years_windiness'],\n", + " x=mc_reg[\"num_years_windiness\"],\n", " xlabel=\"Number of Years in the Windiness Correction\",\n", " ylabel=\"AEP (GWh/yr)\",\n", " ylim=(0, 20),\n", - " plot_kwargs_box={\"flierprops\":dict(marker=\"x\", markeredgecolor=\"tab:blue\")}\n", + " plot_kwargs_box={\"flierprops\": dict(marker=\"x\", markeredgecolor=\"tab:blue\")},\n", ")" ] }, @@ -909,13 +908,13 @@ ], "source": [ "fig, ax, boxes = pa.plot_aep_boxplot(\n", - " x=mc_reg['num_years_windiness'],\n", + " x=mc_reg[\"num_years_windiness\"],\n", " xlabel=\"Number of Years in the Windiness Correction\",\n", " ylim=(0, 20),\n", " figure_kwargs=dict(figsize=(12, 6)),\n", " plot_kwargs_box={\n", " \"flierprops\": dict(marker=\"x\", markeredgecolor=\"tab:blue\"),\n", - " \"medianprops\": dict(linewidth=1.5)\n", + " \"medianprops\": dict(linewidth=1.5),\n", " },\n", " return_fig=True,\n", " with_points=True,\n", diff --git a/sphinx/examples/02b_plant_aep_analysis_cubico.ipynb b/sphinx/examples/02b_plant_aep_analysis_cubico.ipynb index 7ecba5530..12971d42f 100644 --- a/sphinx/examples/02b_plant_aep_analysis_cubico.ipynb +++ b/sphinx/examples/02b_plant_aep_analysis_cubico.ipynb @@ -83,7 +83,7 @@ "metadata": {}, "outputs": [], "source": [ - "asset = \"kelmarsh\" # kelmarsh or penmanshiel\n", + "asset = \"kelmarsh\" # kelmarsh or penmanshiel\n", "\n", "project = project_Cubico.prepare(asset=asset)" ] @@ -125,7 +125,9 @@ ], "source": [ "plot.column_histograms(project.meter.resample(\"1h\").sum(numeric_only=True), columns=[\"MMTR_SupWh\"])\n", - "plot.column_histograms(project.curtail.resample(\"1h\").sum(numeric_only=True), columns=[\"IAVL_DnWh\", \"IAVL_ExtPwrDnWh\"])" + "plot.column_histograms(\n", + " project.curtail.resample(\"1h\").sum(numeric_only=True), columns=[\"IAVL_DnWh\", \"IAVL_ExtPwrDnWh\"]\n", + ")" ] }, { @@ -267,7 +269,7 @@ "metadata": {}, "outputs": [], "source": [ - "pa = MonteCarloAEP(project, reanalysis_products = list(project.reanalysis.keys()))" + "pa = MonteCarloAEP(project, reanalysis_products=list(project.reanalysis.keys()))" ] }, { @@ -728,8 +730,8 @@ "source": [ "pa.plot_normalized_monthly_reanalysis_windspeed(\n", " return_fig=False,\n", - " #xlim=(datetime(2000, 1, 1), datetime(2022, 12, 31)),\n", - " #ylim=(0.8, 1.2),\n", + " # xlim=(datetime(2000, 1, 1), datetime(2022, 12, 31)),\n", + " # ylim=(0.8, 1.2),\n", ")" ] }, @@ -821,15 +823,15 @@ "source": [ "# For both assets there are a few months that aren't representative of long-term losses, so these are excluded.\n", "\n", - "if asset == 'kelmarsh':\n", - " pa.aggregate.loc['2016-02-01',['availability_typical','curtailment_typical']] = False\n", - " pa.aggregate.loc['2016-03-01',['availability_typical','curtailment_typical']] = False\n", - " pa.aggregate.loc['2016-04-01',['availability_typical','curtailment_typical']] = False\n", - " \n", - "elif asset == 'penmanshiel':\n", - " pa.aggregate.loc['2018-01-01',['availability_typical','curtailment_typical']] = False\n", - " pa.aggregate.loc['2018-02-01',['availability_typical','curtailment_typical']] = False\n", - " pa.aggregate.loc['2018-03-01',['availability_typical','curtailment_typical']] = False\n" + "if asset == \"kelmarsh\":\n", + " pa.aggregate.loc[\"2016-02-01\", [\"availability_typical\", \"curtailment_typical\"]] = False\n", + " pa.aggregate.loc[\"2016-03-01\", [\"availability_typical\", \"curtailment_typical\"]] = False\n", + " pa.aggregate.loc[\"2016-04-01\", [\"availability_typical\", \"curtailment_typical\"]] = False\n", + "\n", + "elif asset == \"penmanshiel\":\n", + " pa.aggregate.loc[\"2018-01-01\", [\"availability_typical\", \"curtailment_typical\"]] = False\n", + " pa.aggregate.loc[\"2018-02-01\", [\"availability_typical\", \"curtailment_typical\"]] = False\n", + " pa.aggregate.loc[\"2018-03-01\", [\"availability_typical\", \"curtailment_typical\"]] = False" ] }, { @@ -975,16 +977,18 @@ "outputs": [], "source": [ "# Produce histograms of the various MC-parameters\n", - "mc_reg = pd.DataFrame(data={\n", - " 'slope': pa._mc_slope.ravel(),\n", - " 'intercept': pa._mc_intercept, \n", - " 'num_points': pa._mc_num_points, \n", - " 'metered_energy_fraction': pa.mc_inputs.metered_energy_fraction, \n", - " 'loss_fraction': pa.mc_inputs.loss_fraction,\n", - " 'num_years_windiness': pa.mc_inputs.num_years_windiness, \n", - " 'loss_threshold': pa.mc_inputs.loss_threshold,\n", - " 'reanalysis_product': pa.mc_inputs.reanalysis_product\n", - "})" + "mc_reg = pd.DataFrame(\n", + " data={\n", + " \"slope\": pa._mc_slope.ravel(),\n", + " \"intercept\": pa._mc_intercept,\n", + " \"num_points\": pa._mc_num_points,\n", + " \"metered_energy_fraction\": pa.mc_inputs.metered_energy_fraction,\n", + " \"loss_fraction\": pa.mc_inputs.loss_fraction,\n", + " \"num_years_windiness\": pa.mc_inputs.num_years_windiness,\n", + " \"loss_threshold\": pa.mc_inputs.loss_threshold,\n", + " \"reanalysis_product\": pa.mc_inputs.reanalysis_product,\n", + " }\n", + ")" ] }, { @@ -1043,14 +1047,15 @@ ], "source": [ "# Produce scatter plots of slope and intercept values. Here we focus on only one reanalysis data set\n", - "plt.figure(figsize=(8,6))\n", + "plt.figure(figsize=(8, 6))\n", "plt.plot(\n", - " mc_reg.intercept[mc_reg.reanalysis_product ==list(project.reanalysis.keys())[0]],\n", - " mc_reg.slope[mc_reg.reanalysis_product ==list(project.reanalysis.keys())[0]],\n", - " '.', label=\"Monte Carlo Regression Values\"\n", + " mc_reg.intercept[mc_reg.reanalysis_product == list(project.reanalysis.keys())[0]],\n", + " mc_reg.slope[mc_reg.reanalysis_product == list(project.reanalysis.keys())[0]],\n", + " \".\",\n", + " label=\"Monte Carlo Regression Values\",\n", ")\n", - "plt.xlabel('Intercept (GWh)')\n", - "plt.ylabel('Slope (GWh / (m/s))')\n", + "plt.xlabel(\"Intercept (GWh)\")\n", + "plt.ylabel(\"Slope (GWh / (m/s))\")\n", "plt.legend()\n", "plt.show()" ] @@ -1079,7 +1084,7 @@ } ], "source": [ - "pa.plot_aep_boxplot(x=mc_reg['reanalysis_product'], xlabel=\"Reanalysis Product\")" + "pa.plot_aep_boxplot(x=mc_reg[\"reanalysis_product\"], xlabel=\"Reanalysis Product\")" ] }, { @@ -1110,10 +1115,10 @@ "# NOTE: This is the same method, but calling the same method through the plot module directly\n", "plot.plot_boxplot(\n", " y=pa.results.aep_GWh,\n", - " x=mc_reg['num_years_windiness'],\n", + " x=mc_reg[\"num_years_windiness\"],\n", " xlabel=\"Number of Years in the Windiness Correction\",\n", " ylabel=\"AEP (GWh/yr)\",\n", - " plot_kwargs_box={\"flierprops\":dict(marker=\"x\", markeredgecolor=\"tab:blue\")}\n", + " plot_kwargs_box={\"flierprops\": dict(marker=\"x\", markeredgecolor=\"tab:blue\")},\n", ")" ] }, @@ -1144,12 +1149,12 @@ ], "source": [ "fig, ax, boxes = pa.plot_aep_boxplot(\n", - " x=mc_reg['num_years_windiness'],\n", + " x=mc_reg[\"num_years_windiness\"],\n", " xlabel=\"Number of Years in the Windiness Correction\",\n", " figure_kwargs=dict(figsize=(12, 6)),\n", " plot_kwargs_box={\n", " \"flierprops\": dict(marker=\"x\", markeredgecolor=\"tab:blue\"),\n", - " \"medianprops\": dict(linewidth=1.5)\n", + " \"medianprops\": dict(linewidth=1.5),\n", " },\n", " return_fig=True,\n", " with_points=True,\n", diff --git a/sphinx/examples/02c_augmented_plant_aep_analysis.ipynb b/sphinx/examples/02c_augmented_plant_aep_analysis.ipynb index d0b752fed..285a968b8 100644 --- a/sphinx/examples/02c_augmented_plant_aep_analysis.ipynb +++ b/sphinx/examples/02c_augmented_plant_aep_analysis.ipynb @@ -34,12 +34,6 @@ "metadata": {}, "outputs": [], "source": [ - "import copy\n", - "\n", - "import numpy as np\n", - "import pandas as pd\n", - "import matplotlib.pyplot as plt\n", - "\n", "import project_ENGIE" ] }, diff --git a/sphinx/examples/03_turbine_ideal_energy.ipynb b/sphinx/examples/03_turbine_ideal_energy.ipynb index 102730755..b4f0f5a8e 100644 --- a/sphinx/examples/03_turbine_ideal_energy.ipynb +++ b/sphinx/examples/03_turbine_ideal_energy.ipynb @@ -41,11 +41,8 @@ "source": [ "# Import required packages\n", "import numpy as np\n", - "import pandas as pd\n", - "import matplotlib.pyplot as plt\n", "\n", "from openoa.analysis import TurbineLongTermGrossEnergy\n", - "from openoa.utils import plot\n", "\n", "import project_ENGIE" ] @@ -64,7 +61,7 @@ "outputs": [], "source": [ "# Load plant object and validate for the turbine long term energy analysis type\n", - "project = project_ENGIE.prepare('./data/la_haute_borne', use_cleansed=False)\n", + "project = project_ENGIE.prepare(\"./data/la_haute_borne\", use_cleansed=False)\n", "project.analysis_type.append(\"TurbineLongTermGrossEnergy\")\n", "project.validate()" ] @@ -247,7 +244,7 @@ " UQ=False,\n", " wind_bin_threshold=2.0, # Exclude data outside 2 standard deviations of the median for each power bin\n", " max_power_filter=0.9, # Don't apply bin filter above 0.9 of turbine capacity\n", - " correction_threshold=0.9 # Set the correction threshold to 90%\n", + " correction_threshold=0.9, # Set the correction threshold to 90%\n", ")" ] }, @@ -283,12 +280,12 @@ } ], "source": [ - "# We can choose to save key plots to a file by setting enable_plotting=True and \n", - "# specifying a directory to save the images. For now we turn off this feature. \n", + "# We can choose to save key plots to a file by setting enable_plotting=True and\n", + "# specifying a directory to save the images. For now we turn off this feature.\n", "# ta.run(reanalysis_subset=['era5', 'merra2'], enable_plotting=False, plot_dir=None,\n", "# wind_bin_thresh=wind_bin_thresh, max_power_filter=max_power_filter,\n", "# correction_threshold=correction_threshold)\n", - "ta.run(reanalysis_products=['era5', 'merra2'])" + "ta.run(reanalysis_products=[\"era5\", \"merra2\"])" ] }, { @@ -468,12 +465,18 @@ "metadata": {}, "outputs": [], "source": [ - "ta=TurbineLongTermGrossEnergy(\n", + "ta = TurbineLongTermGrossEnergy(\n", " project,\n", " UQ=True, # enable uncertainty quantification\n", " num_sim=75, # number of Monte Carlo simulations to perform\n", - " wind_bin_threshold=(1, 3), # Data outside of a range of +-1 to +-3 standard deviations from the median for each power bin are discarded\n", - " max_power_filter=(0.8, 0.9), # The bin filter will be applied up to fractions of turbine capacity from 80% to 90%\n", + " wind_bin_threshold=(\n", + " 1,\n", + " 3,\n", + " ), # Data outside of a range of +-1 to +-3 standard deviations from the median for each power bin are discarded\n", + " max_power_filter=(\n", + " 0.8,\n", + " 0.9,\n", + " ), # The bin filter will be applied up to fractions of turbine capacity from 80% to 90%\n", " uncertainty_scada=0.005, # Assumed uncertainty of SCADA power data (0.5%)\n", " correction_threshold=(0.85, 0.95),\n", ")" @@ -509,9 +512,9 @@ } ], "source": [ - "# We can choose to save key plots to a file by setting enable_plotting=True and \n", - "# specifying a directory to save the images. For now we turn off this feature. \n", - "ta.run(reanalysis_products=['era5', 'merra2'])" + "# We can choose to save key plots to a file by setting enable_plotting=True and\n", + "# specifying a directory to save the images. For now we turn off this feature.\n", + "ta.run(reanalysis_products=[\"era5\", \"merra2\"])" ] }, { @@ -549,7 +552,9 @@ "print(f\"Mean long-term turbine ideal energy is {np.mean(ta.plant_gross / 1e6):,.1f} GWh/year\")\n", "\n", "# Uncertainty in long-term annual TIE for whole plant\n", - "print(f\"Uncertainty in long-term turbine ideal energy is {np.std(ta.plant_gross / 1e6):,.1f} GWh/year, or {np.std(ta.plant_gross) / np.mean(ta.plant_gross):.1%} percent\")" + "print(\n", + " f\"Uncertainty in long-term turbine ideal energy is {np.std(ta.plant_gross / 1e6):,.1f} GWh/year, or {np.std(ta.plant_gross) / np.mean(ta.plant_gross):.1%} percent\"\n", + ")" ] }, { diff --git a/sphinx/examples/04_electrical_losses.ipynb b/sphinx/examples/04_electrical_losses.ipynb index dc1ea260c..317e82d6b 100644 --- a/sphinx/examples/04_electrical_losses.ipynb +++ b/sphinx/examples/04_electrical_losses.ipynb @@ -51,8 +51,6 @@ "# Import required packages\n", "from datetime import datetime\n", "\n", - "import numpy as np\n", - "import pandas as pd\n", "\n", "from openoa.analysis import ElectricalLosses\n", "\n", @@ -73,7 +71,7 @@ "outputs": [], "source": [ "# Load wind farm object, append the analysis type for this example, and revalidate the data\n", - "project = project_ENGIE.prepare('./data/la_haute_borne', use_cleansed=False)\n", + "project = project_ENGIE.prepare(\"./data/la_haute_borne\", use_cleansed=False)\n", "project.analysis_type.append(\"ElectricalLosses\")\n", "project.validate()" ] @@ -189,8 +187,7 @@ ], "source": [ "el.plot_monthly_losses(\n", - " xlim=(datetime(month=12, day=1, year=2013), datetime(month=1, day=1, year=2016)),\n", - " ylim=(0, 2.2)\n", + " xlim=(datetime(month=12, day=1, year=2013), datetime(month=1, day=1, year=2016)), ylim=(0, 2.2)\n", ")" ] }, @@ -218,11 +215,15 @@ "source": [ "# Create Electrical Loss object\n", "el = ElectricalLosses(\n", - " project, UQ = True, # enable UQ\n", - " num_sim=3000, # number of Monte Carlo simulations to perform\n", - " uncertainty_meter=0.005, # 0.5% uncertainty in meter data\n", + " project,\n", + " UQ=True, # enable UQ\n", + " num_sim=3000, # number of Monte Carlo simulations to perform\n", + " uncertainty_meter=0.005, # 0.5% uncertainty in meter data\n", " uncertainty_scada=0.005, # 0.5% uncertainty in scada data\n", - " uncertainty_correction_threshold=(0.9, 0.995), # randomly sample between 90% and 99.5% coverage required in a month\n", + " uncertainty_correction_threshold=(\n", + " 0.9,\n", + " 0.995,\n", + " ), # randomly sample between 90% and 99.5% coverage required in a month\n", ")" ] }, diff --git a/sphinx/examples/05_eya_gap_analysis.ipynb b/sphinx/examples/05_eya_gap_analysis.ipynb index 2e2b4e107..481f63b50 100644 --- a/sphinx/examples/05_eya_gap_analysis.ipynb +++ b/sphinx/examples/05_eya_gap_analysis.ipynb @@ -32,7 +32,6 @@ "outputs": [], "source": [ "# Import required packages\n", - "from openoa.utils import plot\n", "\n", "import project_ENGIE\n", "\n", @@ -47,7 +46,7 @@ "outputs": [], "source": [ "# Load plant object and process plant data\n", - "project = project_ENGIE.prepare('./data/la_haute_borne', use_cleansed=False)\n", + "project = project_ENGIE.prepare(\"./data/la_haute_borne\", use_cleansed=False)\n", "\n", "# Add the analysis workflow validations needed below and re-validate\n", "project.analysis_type.extend([\"TurbineLongTermGrossEnergy\", \"ElectricalLosses\"])\n", @@ -78,8 +77,8 @@ ], "source": [ "# Calculate AEP\n", - "pa = project.MonteCarloAEP(reanalysis_products = ['era5', 'merra2'])\n", - "pa.run(num_sim=20000, reanalysis_products=['era5', 'merra2'])" + "pa = project.MonteCarloAEP(reanalysis_products=[\"era5\", \"merra2\"])\n", + "pa.run(num_sim=20000, reanalysis_products=[\"era5\", \"merra2\"])" ] }, { @@ -100,11 +99,11 @@ "ta = project.TurbineLongTermGrossEnergy(\n", " UQ=True,\n", " num_sim=75,\n", - " max_power_filter=(0.8, 0.9), \n", + " max_power_filter=(0.8, 0.9),\n", " wind_bin_threshold=(1.0, 3.0),\n", " correction_threshold=(0.85, 0.95),\n", ")\n", - "ta.run(reanalysis_products=['era5', 'merra2'])" + "ta.run(reanalysis_products=[\"era5\", \"merra2\"])" ] }, { @@ -161,7 +160,7 @@ "aep = pa.results.aep_GWh.mean()\n", "avail = pa.results.avail_pct.mean()\n", "elec = el.electrical_losses[0][0]\n", - "tie = ta.plant_gross[0][0]/1e6\n", + "tie = ta.plant_gross[0][0] / 1e6\n", "\n", "print(f\"AEP = {aep:21.2f} GWh/yr\")\n", "print(f\"Availability Losses = {avail:.2%}\")\n", @@ -180,7 +179,7 @@ " aep=aep, # AEP (GWh/yr)\n", " availability_losses=avail, # Availability loss (fraction)\n", " electrical_losses=elec, # Electrical loss (fraction)\n", - " turbine_ideal_energy=tie # Turbine ideal energy (GWh/yr)\n", + " turbine_ideal_energy=tie, # Turbine ideal energy (GWh/yr)\n", ")\n", "\n", "# Define EYA data (we are fabricating these data as an example)\n", diff --git a/sphinx/examples/06_wake_loss_analysis.ipynb b/sphinx/examples/06_wake_loss_analysis.ipynb index 696769979..bcd892602 100644 --- a/sphinx/examples/06_wake_loss_analysis.ipynb +++ b/sphinx/examples/06_wake_loss_analysis.ipynb @@ -1756,10 +1756,10 @@ "from bokeh.plotting import show\n", "from bokeh.io import output_notebook\n", "from bokeh.resources import INLINE\n", + "\n", "output_notebook(INLINE)\n", "\n", "from openoa.analysis.wake_losses import WakeLosses\n", - "from openoa import PlantData\n", "from openoa.utils import met_data_processing as met\n", "from openoa.utils import plot\n", "\n", @@ -1850,7 +1850,7 @@ } ], "source": [ - "show(plot.plot_windfarm(project.asset,tile_name=\"OpenMap\",plot_width=600,plot_height=600))" + "show(plot.plot_windfarm(project.asset, tile_name=\"OpenMap\", plot_width=600, plot_height=600))" ] }, { @@ -1868,7 +1868,7 @@ "metadata": {}, "outputs": [], "source": [ - "# Modify the SCADA data frame so we can access columns using the (variable, turbine ID) \n", + "# Modify the SCADA data frame so we can access columns using the (variable, turbine ID)\n", "# pair as the column name\n", "scada_df = project.scada.unstack()" ] @@ -2009,12 +2009,17 @@ "source": [ "for ind1, tid1 in enumerate(project.turbine_ids):\n", " for tid2 in [tid for ind, tid in enumerate(project.turbine_ids) if ind > ind1]:\n", - " \n", " # limit to time steps when both turbines are producing power\n", - " valid_indices = (scada_df[(\"WTUR_W\",tid1)] >= 0) & (scada_df[(\"WTUR_W\",tid2)] >= 0)\n", + " valid_indices = (scada_df[(\"WTUR_W\", tid1)] >= 0) & (scada_df[(\"WTUR_W\", tid2)] >= 0)\n", "\n", " plt.figure(figsize=(12, 8))\n", - " plt.plot(scada_df.index[valid_indices],met.wrap_180(scada_df.loc[valid_indices,(\"WMET_HorWdDir\",tid2)].values - scada_df.loc[valid_indices,(\"WMET_HorWdDir\",tid1)].values))\n", + " plt.plot(\n", + " scada_df.index[valid_indices],\n", + " met.wrap_180(\n", + " scada_df.loc[valid_indices, (\"WMET_HorWdDir\", tid2)].values\n", + " - scada_df.loc[valid_indices, (\"WMET_HorWdDir\", tid1)].values\n", + " ),\n", + " )\n", " plt.xlabel(\"Time\")\n", " plt.ylabel(\"Wind Direction Difference (deg)\")\n", " plt.title(f\"{tid2} relative to {tid1}\")" @@ -2115,15 +2120,19 @@ "\n", "for tid1 in turbine_ids_sub:\n", " for tid2 in turbine_ids_sub:\n", - " \n", " # limit to time steps when both turbines are producing power\n", - " valid_indices = (scada_df[(\"WTUR_W\",tid1)] >= 0) & (scada_df[(\"WTUR_W\",tid2)] >= 0)\n", - " \n", + " valid_indices = (scada_df[(\"WTUR_W\", tid1)] >= 0) & (scada_df[(\"WTUR_W\", tid2)] >= 0)\n", + "\n", " # further limit to dates prior to 11/25/2015\n", " valid_indices = valid_indices & (scada_df.index < \"2015-11-25 00:00\")\n", - " \n", - " wind_direction_bias_df.loc[tid1,tid2] = np.mean(met.wrap_180(scada_df.loc[valid_indices, (\"WMET_HorWdDir\",tid2)].values - scada_df.loc[valid_indices, (\"WMET_HorWdDir\",tid1)].values))\n", - " \n", + "\n", + " wind_direction_bias_df.loc[tid1, tid2] = np.mean(\n", + " met.wrap_180(\n", + " scada_df.loc[valid_indices, (\"WMET_HorWdDir\", tid2)].values\n", + " - scada_df.loc[valid_indices, (\"WMET_HorWdDir\", tid1)].values\n", + " )\n", + " )\n", + "\n", "wind_direction_bias_df" ] }, @@ -2172,41 +2181,60 @@ } ], "source": [ - "turbine_id_pairs = [(\"R80711\",\"R80790\"),(\"R80736\",\"R80721\")]\n", + "turbine_id_pairs = [(\"R80711\", \"R80790\"), (\"R80736\", \"R80721\")]\n", "\n", "for tid_up, tid_down in turbine_id_pairs:\n", - "\n", " # limit to time steps when both turbines are producing power\n", - " valid_indices = (scada_df[(\"WTUR_W\",tid_up)] >= 0) & (scada_df[(\"WTUR_W\",tid_down)] >= 0)\n", + " valid_indices = (scada_df[(\"WTUR_W\", tid_up)] >= 0) & (scada_df[(\"WTUR_W\", tid_down)] >= 0)\n", "\n", " # further limit to dates prior to 11/25/2015\n", " valid_indices = valid_indices & (scada_df.index < \"2015-11-25 00:00\")\n", "\n", - " scada_df[(\"wind_direction_bin\",tid_up)] = scada_df[(\"WMET_HorWdDir\",tid_up)].round()\n", + " scada_df[(\"wind_direction_bin\", tid_up)] = scada_df[(\"WMET_HorWdDir\", tid_up)].round()\n", "\n", - " scada_df_bin = scada_df.loc[valid_indices].groupby((\"wind_direction_bin\",tid_up)).mean()\n", + " scada_df_bin = scada_df.loc[valid_indices].groupby((\"wind_direction_bin\", tid_up)).mean()\n", "\n", - " turbine_pair_direction = project.turbine_direction_matrix().loc[tid_down,tid_up]\n", + " turbine_pair_direction = project.turbine_direction_matrix().loc[tid_down, tid_up]\n", "\n", " # ratio between mean power of downstream and upstream turbines vs. wind direction\n", - " power_ratio = scada_df_bin[(\"WTUR_W\",tid_down)]/scada_df_bin[(\"WTUR_W\",tid_up)]\n", - " \n", - " # Find direction where peak wake losses are observed, assuming this direction is within 45 degrees \n", + " power_ratio = scada_df_bin[(\"WTUR_W\", tid_down)] / scada_df_bin[(\"WTUR_W\", tid_up)]\n", + "\n", + " # Find direction where peak wake losses are observed, assuming this direction is within 45 degrees\n", " # of actual direction between turbines\n", - " peak_wake_loss_direction = np.round(turbine_pair_direction) - 45.0 + np.argmin(power_ratio[np.round(turbine_pair_direction)-45:np.round(turbine_pair_direction)+45])\n", + " peak_wake_loss_direction = (\n", + " np.round(turbine_pair_direction)\n", + " - 45.0\n", + " + np.argmin(\n", + " power_ratio[\n", + " np.round(turbine_pair_direction) - 45 : np.round(turbine_pair_direction) + 45\n", + " ]\n", + " )\n", + " )\n", "\n", " # Wind direction offset\n", - " direction_offset = np.round(met.wrap_180(peak_wake_loss_direction - turbine_pair_direction),2)\n", - " \n", - " plt.figure(figsize=(9,6))\n", - " plt.plot(power_ratio,label=\"_nolabel_\")\n", - "\n", - " plt.plot(2*[turbine_pair_direction],[power_ratio.min(),power_ratio.max()],'k--',label = \"Direction between Turbines\")\n", - " plt.plot(2*[peak_wake_loss_direction],[power_ratio.min(),power_ratio.max()],'r--',label = \"Direction of Peak Wake Losses\")\n", + " direction_offset = np.round(met.wrap_180(peak_wake_loss_direction - turbine_pair_direction), 2)\n", + "\n", + " plt.figure(figsize=(9, 6))\n", + " plt.plot(power_ratio, label=\"_nolabel_\")\n", + "\n", + " plt.plot(\n", + " 2 * [turbine_pair_direction],\n", + " [power_ratio.min(), power_ratio.max()],\n", + " \"k--\",\n", + " label=\"Direction between Turbines\",\n", + " )\n", + " plt.plot(\n", + " 2 * [peak_wake_loss_direction],\n", + " [power_ratio.min(), power_ratio.max()],\n", + " \"r--\",\n", + " label=\"Direction of Peak Wake Losses\",\n", + " )\n", " plt.legend()\n", " plt.xlabel(\"Wind Direction (deg)\")\n", " plt.ylabel(\"$P_{down}/P_{up}$ (-)\")\n", - " plt.title(f\"Upstream Turbine = {tid_up}, Downstream Turbine = {tid_down}. Wind Direction Offset = {direction_offset} deg.\")" + " plt.title(\n", + " f\"Upstream Turbine = {tid_up}, Downstream Turbine = {tid_down}. Wind Direction Offset = {direction_offset} deg.\"\n", + " )" ] }, { @@ -2257,9 +2285,9 @@ " wind_direction_asset_ids=[\"R80711\", \"R80721\", \"R80736\"],\n", " start_date=None,\n", " end_date=\"2015-11-25 00:00\",\n", - " reanalysis_products=[\"merra2\",\"era5\"],\n", + " reanalysis_products=[\"merra2\", \"era5\"],\n", " end_date_lt=None,\n", - " UQ=False\n", + " UQ=False,\n", ")" ] }, @@ -2316,7 +2344,7 @@ " ws_bin_width_LT_corr=1.0,\n", " num_years_LT=20,\n", " assume_no_wakes_high_ws_LT_corr=True,\n", - " no_wakes_ws_thresh_LT_corr=15.0\n", + " no_wakes_ws_thresh_LT_corr=15.0,\n", ")" ] }, @@ -2452,10 +2480,14 @@ ], "source": [ "for i in range(len(wl.turbine_ids)):\n", - " print(f\"Period-of-record wake losses for turbine {wl.turbine_ids[i]}: {np.round(100*wl.turbine_wake_losses_por[i],2)}%\")\n", - " \n", + " print(\n", + " f\"Period-of-record wake losses for turbine {wl.turbine_ids[i]}: {np.round(100 * wl.turbine_wake_losses_por[i], 2)}%\"\n", + " )\n", + "\n", "for i in range(len(wl.turbine_ids)):\n", - " print(f\"Long-term corrected wake losses for turbine {wl.turbine_ids[i]}: {np.round(100*wl.turbine_wake_losses_lt[i],2)}%\")" + " print(\n", + " f\"Long-term corrected wake losses for turbine {wl.turbine_ids[i]}: {np.round(100 * wl.turbine_wake_losses_lt[i], 2)}%\"\n", + " )" ] }, { @@ -2482,7 +2514,7 @@ } ], "source": [ - "axes = wl.plot_wake_losses_by_wind_direction(turbine_id = \"R80711\")" + "axes = wl.plot_wake_losses_by_wind_direction(turbine_id=\"R80711\")" ] }, { @@ -2509,7 +2541,7 @@ } ], "source": [ - "axes = wl.plot_wake_losses_by_wind_direction(turbine_id = \"R80721\")" + "axes = wl.plot_wake_losses_by_wind_direction(turbine_id=\"R80721\")" ] }, { @@ -2538,7 +2570,7 @@ } ], "source": [ - "axes = wl.plot_wake_losses_by_wind_speed(turbine_id = \"R80711\")" + "axes = wl.plot_wake_losses_by_wind_speed(turbine_id=\"R80711\")" ] }, { @@ -2565,7 +2597,7 @@ } ], "source": [ - "axes = wl.plot_wake_losses_by_wind_speed(turbine_id = \"R80721\")" + "axes = wl.plot_wake_losses_by_wind_speed(turbine_id=\"R80721\")" ] }, { @@ -2597,9 +2629,9 @@ " wind_direction_asset_ids=[\"R80711\", \"R80721\", \"R80736\"],\n", " start_date=None,\n", " end_date=\"2015-11-25 00:00\",\n", - " reanalysis_products=[\"merra2\",\"era5\"],\n", + " reanalysis_products=[\"merra2\", \"era5\"],\n", " end_date_lt=None,\n", - " UQ=True\n", + " UQ=True,\n", ")" ] }, @@ -2641,7 +2673,7 @@ " ws_bin_width_LT_corr=1.0,\n", " num_years_LT=(10, 20),\n", " assume_no_wakes_high_ws_LT_corr=True,\n", - " no_wakes_ws_thresh_LT_corr=15.0\n", + " no_wakes_ws_thresh_LT_corr=15.0,\n", ")" ] }, @@ -2683,8 +2715,12 @@ "print(f\"Standard deviation of period-of-record wake losses: {wl_uq.wake_losses_por_std:.2%}\")\n", "print(f\"Standard deviation of long-term corrected wake losses: {wl_uq.wake_losses_lt_std:.2%}\")\n", "\n", - "print(f\"95% confidence interval of period-of-record wake losses: [{np.percentile(wl_uq.wake_losses_por,2.5):.2%}, {np.percentile(wl_uq.wake_losses_por,97.5):.2%}]\")\n", - "print(f\"95% confidence interval of long-term corrected wake losses: [{np.percentile(wl_uq.wake_losses_por,2.5):.2%}, {np.percentile(wl_uq.wake_losses_lt,97.5):.2%}]\")" + "print(\n", + " f\"95% confidence interval of period-of-record wake losses: [{np.percentile(wl_uq.wake_losses_por, 2.5):.2%}, {np.percentile(wl_uq.wake_losses_por, 97.5):.2%}]\"\n", + ")\n", + "print(\n", + " f\"95% confidence interval of long-term corrected wake losses: [{np.percentile(wl_uq.wake_losses_por, 2.5):.2%}, {np.percentile(wl_uq.wake_losses_lt, 97.5):.2%}]\"\n", + ")" ] }, { @@ -2795,11 +2831,19 @@ "for i in range(len(wl_uq.turbine_ids)):\n", " print(f\"Turbine {wl_uq.turbine_ids[i]}:\")\n", " print(f\"Mean period-of-record wake losses: {wl_uq.turbine_wake_losses_por_mean[i]:.2%}\")\n", - " print(f\"Standard deviation of period-of-record wake losses: {wl_uq.turbine_wake_losses_por_std[i]:.2%}\")\n", - " print(f\"95% confidence interval of period-of-record wake losses: [{np.percentile(wl_uq.turbine_wake_losses_por[:,i],2.5):.2%}, {np.percentile(wl_uq.turbine_wake_losses_por[:,i],97.5):.2%}]\")\n", + " print(\n", + " f\"Standard deviation of period-of-record wake losses: {wl_uq.turbine_wake_losses_por_std[i]:.2%}\"\n", + " )\n", + " print(\n", + " f\"95% confidence interval of period-of-record wake losses: [{np.percentile(wl_uq.turbine_wake_losses_por[:, i], 2.5):.2%}, {np.percentile(wl_uq.turbine_wake_losses_por[:, i], 97.5):.2%}]\"\n", + " )\n", " print(f\"Mean long-term corrected wake losses: {wl_uq.turbine_wake_losses_lt_mean[i]:.2%}\")\n", - " print(f\"Standard deviation of long-term corrected wake losses: {wl_uq.turbine_wake_losses_lt_std[i]:.2%}\")\n", - " print(f\"95% confidence interval of long-term corrected wake losses: [{np.percentile(wl_uq.turbine_wake_losses_lt[:,i],2.5):.2%}, {np.percentile(wl_uq.turbine_wake_losses_lt[:,i],97.5):.2%}]\\n\")" + " print(\n", + " f\"Standard deviation of long-term corrected wake losses: {wl_uq.turbine_wake_losses_lt_std[i]:.2%}\"\n", + " )\n", + " print(\n", + " f\"95% confidence interval of long-term corrected wake losses: [{np.percentile(wl_uq.turbine_wake_losses_lt[:, i], 2.5):.2%}, {np.percentile(wl_uq.turbine_wake_losses_lt[:, i], 97.5):.2%}]\\n\"\n", + " )" ] }, { @@ -2846,7 +2890,7 @@ } ], "source": [ - "axes=wl_uq.plot_wake_losses_by_wind_direction(turbine_id=\"R80721\")" + "axes = wl_uq.plot_wake_losses_by_wind_direction(turbine_id=\"R80721\")" ] }, { @@ -3055,7 +3099,7 @@ "source": [ "plt.figure(figsize=(10, 7))\n", "for id in project.turbine_ids:\n", - " plt.plot(df_speedup[\"wd\"],df_speedup[id],label=id)\n", + " plt.plot(df_speedup[\"wd\"], df_speedup[id], label=id)\n", "plt.xlabel(\"Wind Direction (deg)\")\n", "plt.ylabel(\"Relative Speedup Factor (-)\")\n", "plt.legend()" @@ -3096,11 +3140,11 @@ " wind_direction_asset_ids=[\"R80711\", \"R80721\", \"R80736\"],\n", " start_date=None,\n", " end_date=\"2015-11-25 00:00\",\n", - " reanalysis_products=[\"merra2\",\"era5\"],\n", + " reanalysis_products=[\"merra2\", \"era5\"],\n", " end_date_lt=None,\n", " UQ=False,\n", " correct_for_ws_heterogeneity=True,\n", - " ws_speedup_factor_map=\"example_la_haute_borne_ws_speedup_factors.csv\"\n", + " ws_speedup_factor_map=\"example_la_haute_borne_ws_speedup_factors.csv\",\n", ")" ] }, @@ -3135,7 +3179,7 @@ " ws_bin_width_LT_corr=1.0,\n", " num_years_LT=20,\n", " assume_no_wakes_high_ws_LT_corr=True,\n", - " no_wakes_ws_thresh_LT_corr=15.0\n", + " no_wakes_ws_thresh_LT_corr=15.0,\n", ")" ] }, @@ -3165,8 +3209,12 @@ "source": [ "print(f\"Basic period-of-record wake losses: {wl.wake_losses_por:.2%}\")\n", "print(f\"Basic long-term corrected wake losses: {wl.wake_losses_lt:.2%}\")\n", - "print(f\"Period-of-record wake losses with wind speed heterogeneity corrections: {wl_het.wake_losses_por:.2%}\")\n", - "print(f\"Long-term corrected wake losses with wind speed heterogeneity corrections: {wl_het.wake_losses_lt:.2%}\")" + "print(\n", + " f\"Period-of-record wake losses with wind speed heterogeneity corrections: {wl_het.wake_losses_por:.2%}\"\n", + ")\n", + "print(\n", + " f\"Long-term corrected wake losses with wind speed heterogeneity corrections: {wl_het.wake_losses_lt:.2%}\"\n", + ")" ] }, { @@ -3290,18 +3338,26 @@ ], "source": [ "for i in range(len(wl.turbine_ids)):\n", - " print(f\"Basic period-of-record wake losses for turbine {wl.turbine_ids[i]}: {np.round(100*wl.turbine_wake_losses_por[i],2)}%\")\n", - " \n", + " print(\n", + " f\"Basic period-of-record wake losses for turbine {wl.turbine_ids[i]}: {np.round(100 * wl.turbine_wake_losses_por[i], 2)}%\"\n", + " )\n", + "\n", "for i in range(len(wl.turbine_ids)):\n", - " print(f\"Basic long-term corrected wake losses for turbine {wl.turbine_ids[i]}: {np.round(100*wl.turbine_wake_losses_lt[i],2)}%\")\n", + " print(\n", + " f\"Basic long-term corrected wake losses for turbine {wl.turbine_ids[i]}: {np.round(100 * wl.turbine_wake_losses_lt[i], 2)}%\"\n", + " )\n", "\n", "print(\"\")\n", "\n", "for i in range(len(wl_het.turbine_ids)):\n", - " print(f\"Period-of-record wake losses for turbine {wl_het.turbine_ids[i]} with wind speed heterogeneity corrections: {np.round(100*wl_het.turbine_wake_losses_por[i],2)}%\")\n", - " \n", + " print(\n", + " f\"Period-of-record wake losses for turbine {wl_het.turbine_ids[i]} with wind speed heterogeneity corrections: {np.round(100 * wl_het.turbine_wake_losses_por[i], 2)}%\"\n", + " )\n", + "\n", "for i in range(len(wl_het.turbine_ids)):\n", - " print(f\"Long-term corrected wake losses for turbine {wl_het.turbine_ids[i]} with wind speed heterogeneity corrections: {np.round(100*wl_het.turbine_wake_losses_lt[i],2)}%\")" + " print(\n", + " f\"Long-term corrected wake losses for turbine {wl_het.turbine_ids[i]} with wind speed heterogeneity corrections: {np.round(100 * wl_het.turbine_wake_losses_lt[i], 2)}%\"\n", + " )" ] }, { @@ -3342,8 +3398,8 @@ } ], "source": [ - "fig, axes = wl.plot_wake_losses_by_wind_direction(turbine_id = \"R80711\", return_fig=True)\n", - "axes[0].set_title(axes[0].get_title()+\", using Basic Wake Loss Estimation Method\")" + "fig, axes = wl.plot_wake_losses_by_wind_direction(turbine_id=\"R80711\", return_fig=True)\n", + "axes[0].set_title(axes[0].get_title() + \", using Basic Wake Loss Estimation Method\")" ] }, { @@ -3373,8 +3429,8 @@ } ], "source": [ - "fig, axes = wl_het.plot_wake_losses_by_wind_direction(turbine_id = \"R80711\", return_fig=True)\n", - "axes[0].set_title(axes[0].get_title()+\", with Wind Speed Heterogeneity Corrections\")" + "fig, axes = wl_het.plot_wake_losses_by_wind_direction(turbine_id=\"R80711\", return_fig=True)\n", + "axes[0].set_title(axes[0].get_title() + \", with Wind Speed Heterogeneity Corrections\")" ] }, { @@ -3413,8 +3469,8 @@ } ], "source": [ - "fig, axes = wl.plot_wake_losses_by_wind_direction(turbine_id = \"R80721\", return_fig=True)\n", - "axes[0].set_title(axes[0].get_title()+\", using Basic Wake Loss Estimation Method\")" + "fig, axes = wl.plot_wake_losses_by_wind_direction(turbine_id=\"R80721\", return_fig=True)\n", + "axes[0].set_title(axes[0].get_title() + \", using Basic Wake Loss Estimation Method\")" ] }, { @@ -3444,8 +3500,8 @@ } ], "source": [ - "fig, axes = wl_het.plot_wake_losses_by_wind_direction(turbine_id = \"R80721\", return_fig=True)\n", - "axes[0].set_title(axes[0].get_title()+\", with Wind Speed Heterogeneity Corrections\")" + "fig, axes = wl_het.plot_wake_losses_by_wind_direction(turbine_id=\"R80721\", return_fig=True)\n", + "axes[0].set_title(axes[0].get_title() + \", with Wind Speed Heterogeneity Corrections\")" ] }, { diff --git a/sphinx/examples/07_static_yaw_misalignment.ipynb b/sphinx/examples/07_static_yaw_misalignment.ipynb index 035d4537a..c8d57311d 100644 --- a/sphinx/examples/07_static_yaw_misalignment.ipynb +++ b/sphinx/examples/07_static_yaw_misalignment.ipynb @@ -1803,16 +1803,14 @@ "source": [ "# Import required packages\n", "import numpy as np\n", - "import pandas as pd\n", - "from matplotlib import pyplot as plt\n", "\n", "from bokeh.plotting import show\n", "from bokeh.io import output_notebook\n", "from bokeh.resources import INLINE\n", + "\n", "output_notebook(INLINE)\n", "\n", "from openoa.analysis.yaw_misalignment import StaticYawMisalignment\n", - "from openoa import PlantData\n", "from openoa.utils import plot, filters\n", "\n", "import project_ENGIE\n", @@ -1902,7 +1900,7 @@ } ], "source": [ - "show(plot.plot_windfarm(project.asset,tile_name=\"OpenMap\",plot_width=600,plot_height=600))" + "show(plot.plot_windfarm(project.asset, tile_name=\"OpenMap\", plot_width=600, plot_height=600))" ] }, { @@ -1941,7 +1939,7 @@ " xlabel=\"Wind Speed (m/s)\",\n", " ylabel=\"Blade Pitch Angle (deg.)\",\n", " xlim=(0, 15),\n", - " ylim=(-2,3),\n", + " ylim=(-2, 3),\n", " max_cols=2,\n", " figure_kwargs={\"figsize\": (12, 8)},\n", ")" @@ -2007,14 +2005,13 @@ "power_bin_mad_thresh = 7.0\n", "\n", "for t in project.turbine_ids:\n", - " \n", " # TODO: apply bin power curve filtering\n", - " df_sub = project.scada.loc[(slice(None), t),:]\n", + " df_sub = project.scada.loc[(slice(None), t), :]\n", " df_sub = df_sub.loc[df_sub[\"WROT_BlPthAngVal\"] <= pitch_threshold]\n", - " \n", + "\n", " # Apply power bin filter\n", - " turb_capac=project.asset.loc[t, \"rated_power\"]\n", - " flag_bin=filters.bin_filter(\n", + " turb_capac = project.asset.loc[t, \"rated_power\"]\n", + " flag_bin = filters.bin_filter(\n", " bin_col=df_sub[\"WTUR_W\"],\n", " value_col=df_sub[\"WMET_HorWdSpd\"],\n", " bin_width=0.04 * 0.94 * turb_capac,\n", @@ -2031,7 +2028,7 @@ " power=df_sub[\"WTUR_W\"],\n", " flag=flag_bin,\n", " flag_labels=(\"Outliers\", \"Power Curve\"),\n", - " legend=True\n", + " legend=True,\n", " )" ] }, @@ -2057,11 +2054,7 @@ "metadata": {}, "outputs": [], "source": [ - "yaw_mis = StaticYawMisalignment(\n", - " plant=project,\n", - " turbine_ids=None,\n", - " UQ=False\n", - ")" + "yaw_mis = StaticYawMisalignment(plant=project, turbine_ids=None, UQ=False)" ] }, { @@ -2106,7 +2099,7 @@ " pitch_thresh=pitch_threshold,\n", " max_power_filter=0.95,\n", " power_bin_mad_thresh=power_bin_mad_thresh,\n", - " use_power_coeff=True\n", + " use_power_coeff=True,\n", ")" ] }, @@ -2137,7 +2130,9 @@ ], "source": [ "for i, t in enumerate(yaw_mis.turbine_ids):\n", - " print(f\"Overall yaw misalignment for Turbine {t}: {np.round(yaw_mis.yaw_misalignment[i],1)} degrees\")" + " print(\n", + " f\"Overall yaw misalignment for Turbine {t}: {np.round(yaw_mis.yaw_misalignment[i], 1)} degrees\"\n", + " )" ] }, { @@ -2205,7 +2200,7 @@ } ], "source": [ - "axes_dict = yaw_mis.plot_yaw_misalignment_by_turbine(return_fig = True)" + "axes_dict = yaw_mis.plot_yaw_misalignment_by_turbine(return_fig=True)" ] }, { @@ -2234,11 +2229,7 @@ "metadata": {}, "outputs": [], "source": [ - "yaw_mis = StaticYawMisalignment(\n", - " plant=project,\n", - " turbine_ids=None,\n", - " UQ=True\n", - ")" + "yaw_mis = StaticYawMisalignment(plant=project, turbine_ids=None, UQ=True)" ] }, { @@ -2280,7 +2271,7 @@ " pitch_thresh=pitch_threshold,\n", " max_power_filter=(0.92, 0.98),\n", " power_bin_mad_thresh=(4.0, 10.0),\n", - " use_power_coeff=True\n", + " use_power_coeff=True,\n", ")" ] }, @@ -2321,9 +2312,15 @@ ], "source": [ "for i, t in enumerate(yaw_mis.turbine_ids):\n", - " print(f\"Mean overall yaw misalignment for Turbine {t}: {np.round(yaw_mis.yaw_misalignment_avg[i],1)} degrees\")\n", - " print(f\"Std. Dev. of overall yaw misalignment for Turbine {t}: {np.round(yaw_mis.yaw_misalignment_std[i],1)} degrees\")\n", - " print(f\"95% confidence interval of overall yaw misalignment for Turbine {t}: [{np.round(yaw_mis.yaw_misalignment_95ci[i,0],1)} deg., {np.round(yaw_mis.yaw_misalignment_95ci[i,1],1)} deg.]\")\n" + " print(\n", + " f\"Mean overall yaw misalignment for Turbine {t}: {np.round(yaw_mis.yaw_misalignment_avg[i], 1)} degrees\"\n", + " )\n", + " print(\n", + " f\"Std. Dev. of overall yaw misalignment for Turbine {t}: {np.round(yaw_mis.yaw_misalignment_std[i], 1)} degrees\"\n", + " )\n", + " print(\n", + " f\"95% confidence interval of overall yaw misalignment for Turbine {t}: [{np.round(yaw_mis.yaw_misalignment_95ci[i, 0], 1)} deg., {np.round(yaw_mis.yaw_misalignment_95ci[i, 1], 1)} deg.]\"\n", + " )" ] }, { @@ -2392,7 +2389,7 @@ } ], "source": [ - "axes_dict = yaw_mis.plot_yaw_misalignment_by_turbine(return_fig = True)" + "axes_dict = yaw_mis.plot_yaw_misalignment_by_turbine(return_fig=True)" ] }, { diff --git a/test/conftest.py b/test/conftest.py index 5e645f12b..0d876f3a1 100644 --- a/test/conftest.py +++ b/test/conftest.py @@ -1,11 +1,11 @@ import sys from pathlib import Path -import pytest examples_folder = Path(__file__).resolve().parents[1] sys.path.append(examples_folder) -from examples import project_ENGIE, example_data_path_str # noqa: disable=E402 +from examples import project_ENGIE, example_data_path_str # ruff: ignore[E402,F401] + ROOT = Path(__file__).parent diff --git a/test/regression/electrical_losses.py b/test/regression/electrical_losses.py index c91455d05..e300eeb85 100644 --- a/test/regression/electrical_losses.py +++ b/test/regression/electrical_losses.py @@ -5,6 +5,7 @@ from openoa.analysis.electrical_losses import ElectricalLosses + from test.conftest import project_ENGIE, example_data_path_str # isort: skip diff --git a/test/regression/eya_gap_analysis.py b/test/regression/eya_gap_analysis.py index 73d3e90ea..ee1e457fd 100644 --- a/test/regression/eya_gap_analysis.py +++ b/test/regression/eya_gap_analysis.py @@ -3,7 +3,6 @@ import numpy as np import numpy.testing as npt -from openoa.analysis.eya_gap_analysis import EYAGapAnalysis from test.conftest import project_ENGIE, example_data_path_str # isort: skip @@ -12,24 +11,24 @@ class EYAGAPAnalysis(unittest.TestCase): def setUp(self): np.random.seed(42) # Set up operational results data - oa_data = dict( - aep=448.0, - availability_losses=0.0493, - electrical_losses=0.012, - turbine_ideal_energy=477.8, - ) + oa_data = { + "aep": 448.0, + "availability_losses": 0.0493, + "electrical_losses": 0.012, + "turbine_ideal_energy": 477.8, + } # AEP (GWh/yr), availability loss (fraction), electrical loss (fraction), turbine ideal energy (GWh/yr) # Set up EYA estimates - eya_data = dict( - aep=467.0, - gross_energy=597.14, - availability_losses=0.062, - electrical_losses=0.024, - turbine_losses=0.037, - blade_degradation_losses=0.011, - wake_losses=0.087, - ) + eya_data = { + "aep": 467.0, + "gross_energy": 597.14, + "availability_losses": 0.062, + "electrical_losses": 0.024, + "turbine_losses": 0.037, + "blade_degradation_losses": 0.011, + "wake_losses": 0.087, + } self.project = project_ENGIE.prepare(example_data_path_str, use_cleansed=False) diff --git a/test/regression/long_term_monte_carlo_aep.py b/test/regression/long_term_monte_carlo_aep.py index 08a142dee..d38982779 100644 --- a/test/regression/long_term_monte_carlo_aep.py +++ b/test/regression/long_term_monte_carlo_aep.py @@ -9,6 +9,7 @@ from openoa.analysis import MonteCarloAEP + from test.conftest import project_ENGIE, example_data_path_str # isort: skip @@ -19,8 +20,7 @@ def reset_prng(): class TestLongTermMonteCarloAEP(unittest.TestCase): def setUp(self): - """ - Python Unittest setUp method. + """Python Unittest setUp method. Load data from disk into PlantData objects and prepare the data for testing the AEP method. """ reset_prng() @@ -39,9 +39,7 @@ def setUp(self): ] def test_monthly_inputs(self): - """ - Test inputs to the regression model, at monthly time resolution - """ + """Test inputs to the regression model, at monthly time resolution.""" reset_prng() self.analysis = MonteCarloAEP( @@ -60,9 +58,7 @@ def test_monthly_inputs(self): self.check_process_reanalysis_data_monthly(df, df_rean) def test_reanalysis_aggregate_monthly(self): - """ - Test reanalysis start and end dates depending on time resolution and end date argument - """ + """Test reanalysis start and end dates depending on time resolution and end date argument.""" reset_prng() # ____________________________________________________________________ # Test default aggregate reanalysis values and date range, at monthly time resolution @@ -81,10 +77,10 @@ def test_reanalysis_aggregate_monthly(self): expected = {"merra2": [7.584891, 8.679547], "era5": [7.241081, 8.188632]} computed = { key: df_rean.loc[[df_rean.index[0], df_rean.index[-1]], key].to_numpy() - for key in expected.keys() + for key in expected } - for key in expected.keys(): + for key in expected: nptest.assert_array_almost_equal(expected[key], computed[key]) # ____________________________________________________________________ @@ -127,10 +123,10 @@ def test_reanalysis_aggregate_monthly(self): expected = {"merra2": [7.584891, 6.529796], "era5": [7.241081, 6.644804]} computed = { key: df_rean.loc[[df_rean.index[0], df_rean.index[-1]], key].to_numpy() - for key in expected.keys() + for key in expected } - for key in expected.keys(): + for key in expected: nptest.assert_array_almost_equal(expected[key], computed[key]) def test_reanalysis_aggregate_daily(self): @@ -152,10 +148,10 @@ def test_reanalysis_aggregate_daily(self): expected = {"merra2": [12.868168, 5.152958], "era5": [12.461761, 5.238968]} computed = { key: df_rean.loc[[df_rean.index[0], df_rean.index[-1]], key].to_numpy() - for key in expected.keys() + for key in expected } - for key in expected.keys(): + for key in expected: nptest.assert_array_almost_equal(expected[key], computed[key]) # ____________________________________________________________________ @@ -198,10 +194,10 @@ def test_reanalysis_aggregate_daily(self): expected = {"merra2": [12.868168, 14.571084], "era5": [12.461761, 14.045798]} computed = { key: df_rean.loc[[df_rean.index[0], df_rean.index[-1]], key].to_numpy() - for key in expected.keys() + for key in expected } - for key in expected.keys(): + for key in expected: nptest.assert_array_almost_equal(expected[key], computed[key]) def test_reanalysis_aggregate_hourly(self): @@ -223,10 +219,10 @@ def test_reanalysis_aggregate_hourly(self): expected = {"merra2": [10.509840, 9.096710], "era5": [9.202639, 9.486806]} computed = { key: df_rean.loc[[df_rean.index[0], df_rean.index[-1]], key].to_numpy() - for key in expected.keys() + for key in expected } - for key in expected.keys(): + for key in expected: nptest.assert_array_almost_equal(expected[key], computed[key]) # ____________________________________________________________________ @@ -269,10 +265,10 @@ def test_reanalysis_aggregate_hourly(self): expected = {"merra2": [10.509840, 16.985526], "era5": [9.202639, 15.608469]} computed = { key: df_rean.loc[[df_rean.index[0], df_rean.index[-1]], key].to_numpy() - for key in expected.keys() + for key in expected } - for key in expected.keys(): + for key in expected: nptest.assert_array_almost_equal(expected[key], computed[key]) def test_monthly_lin(self): @@ -399,7 +395,7 @@ def check_process_loss_estimates_monthly(self, df): # Availablity, curtailment nan fields both 0, NaN flag is all False nptest.assert_array_equal(df["avail_nan_perc"].values, np.repeat(0.0, df.shape[0])) nptest.assert_array_equal(df["curt_nan_perc"].values, np.repeat(0.0, df.shape[0])) - nptest.assert_array_equal(df["nan_flag"].values, np.repeat(False, df.shape[0])) + nptest.assert_array_equal(df["nan_flag"].values, np.repeat(False, df.shape[0])) # noqa: FBT003 # Check a few reported availabilty and curtailment values expected_avail_gwh = pd.Series([0.029417, 0.021005, 0.000444]) @@ -425,11 +421,9 @@ def check_process_reanalysis_data_monthly(self, df, df_rean): } date_ind = pd.to_datetime(["2014-06-01", "2014-12-01", "2015-10-01"]) - computed = {key: df.loc[date_ind, key].to_numpy() for key in expected.keys()} + computed = {key: df.loc[date_ind, key].to_numpy() for key in expected} - print(computed) - - for key in expected.keys(): + for key in expected: nptest.assert_array_almost_equal(expected[key], computed[key]) # check that date range is truncated correctly @@ -453,7 +447,7 @@ def check_process_loss_estimates_daily(self, df): # Availablity, curtailment nan fields both 0, NaN flag is all False nptest.assert_array_equal(df["avail_nan_perc"].values, np.repeat(0.0, df.shape[0])) nptest.assert_array_equal(df["curt_nan_perc"].values, np.repeat(0.0, df.shape[0])) - nptest.assert_array_equal(df["nan_flag"].values, np.repeat(False, df.shape[0])) + nptest.assert_array_equal(df["nan_flag"].values, np.repeat(False, df.shape[0])) # noqa: FBT003 # Check a few reported availabilty and curtailment values expected_avail_gwh = pd.Series([0.0000483644, 0.000000, 0.000000]) @@ -479,11 +473,9 @@ def check_process_reanalysis_data_daily(self, df): } date_ind = pd.to_datetime(["2014-01-02", "2014-10-12", "2015-12-28"]) - computed = {key: df.loc[date_ind, key].to_numpy() for key in expected.keys()} - - print(computed) + computed = {key: df.loc[date_ind, key].to_numpy() for key in expected} - for key in expected.keys(): + for key in expected: with self.subTest(f"checking {key}"): nptest.assert_array_almost_equal(expected[key], computed[key]) diff --git a/test/regression/plantdata_validations.py b/test/regression/plantdata_validations.py index a664a9fd7..9b0154423 100644 --- a/test/regression/plantdata_validations.py +++ b/test/regression/plantdata_validations.py @@ -4,23 +4,18 @@ from pathlib import Path import yaml -import pytest -from numpy.testing import assert_array_equal from pandas.testing import assert_frame_equal from openoa import PlantData from openoa.schema import ANALYSIS_REQUIREMENTS, ReanalysisMetaData from openoa.schema.schema import create_schema, create_analysis_schema + from test.conftest import project_ENGIE, example_data_path_str # isort: skip class TestPlantData(unittest.TestCase): - """ - TestPlantData - - Tests the construction, validation, and methods of PlantData using La Haute Born wind plant data - """ + """Tests the construction, validation, and methods of PlantData using La Haute Born wind plant data.""" @classmethod def setUpClass(cls): @@ -35,9 +30,7 @@ def setUpClass(cls): ) def setUp(self): - """ - Create the plantdata object - """ + """Create the plantdata object.""" self.plant = PlantData( analysis_type=None, # No validation desired at this point in time metadata=example_data_path_str + "/../plant_meta.yml", @@ -49,10 +42,8 @@ def setUp(self): ) def test_analysis_type_values(self): - """ - Test the acceptance of the valid inputs, except None - """ - valid = [*ANALYSIS_REQUIREMENTS] + ["all", None] + """Test the acceptance of the valid inputs, except None.""" + valid = [*ANALYSIS_REQUIREMENTS, "all", None] self.plant.analysis_type = valid self.assertTrue(self.plant.analysis_type == valid) @@ -67,9 +58,7 @@ def test_analysis_type_values(self): self.plant.analysis_type = "this is wrong" def test_validatePlantForAEP(self): - """ - The example plant should validate for MonteCarloAEP analysis type - """ + """The example plant should validate for MonteCarloAEP analysis type.""" self.plant.analysis_type = "MonteCarloAEP" self.plant.validate() @@ -77,17 +66,13 @@ def test_validatePlantForAEP(self): assert None not in self.plant.analysis_type def test_doesNotValidateForAll(self): - """ - The example plant should not validate for MonteCarloAEP analysis type - """ + """The example plant should not validate for MonteCarloAEP analysis type.""" with self.assertRaises(ValueError): self.plant.analysis_type = "all" self.plant.validate() def test_update_columns(self): - """ - Tests that the column names are successfully mapped to the standardized names. - """ + """Tests that the column names are successfully mapped to the standardized names.""" # Put the plant analysis type back in working order self.plant.analysis_type = "MonteCarloAEP" self.plant.validate() @@ -112,9 +97,7 @@ def test_update_columns(self): assert len(re_original.intersection(self.plant.reanalysis[name].columns)) == 0 def test_toCSV(self): - """ - Save this plant to a temporary directory, load it in, and make sure the data matches. - """ + """Save this plant to a temporary directory, load it in, and make sure the data matches.""" # Save data_path = tempfile.mkdtemp() self.plant.to_csv(save_path=data_path, with_openoa_col_names=True) @@ -165,11 +148,7 @@ def test_toCSV(self): class TestPlantDatPartial(unittest.TestCase): - """ - TestPlantData - - Tests the construction, validation, and methods of PlantData using La Haute Born wind plant data - """ + """Tests the construction, validation, and methods of PlantData using La Haute Born wind plant data.""" @classmethod def setUpClass(cls): @@ -184,10 +163,8 @@ def setUpClass(cls): ) def setUp(self): - """ - Create the plantdata object - """ - with open(example_data_path_str + "/../plant_meta.yml") as f: + """Create the plantdata object.""" + with Path(example_data_path_str + "/../plant_meta.yml").open() as f: meta_partial = yaml.safe_load(f) meta_partial.pop("reanalysis") self.plant = PlantData( @@ -206,7 +183,7 @@ def test_reanalysis_missing_metadata(self): """Tests that when there are missing products in the reanalysis metadata, that a KeyError is raised early. """ - with open(example_data_path_str + "/../plant_meta.yml") as f: + with Path(example_data_path_str + "/../plant_meta.yml").open() as f: metadata = yaml.safe_load(f) # Raised when all missing @@ -226,19 +203,19 @@ class TestSchema(unittest.TestCase): def setUp(self): schema_path = Path(__file__).resolve().parents[2] / "openoa/schema" - with open(schema_path / "full_schema.yml") as f: + with (schema_path / "full_schema.yml").open() as f: self.full_schema = yaml.safe_load(f) - with open(schema_path / "base_electrical_losses_schema.yml") as f: + with (schema_path / "base_electrical_losses_schema.yml").open() as f: self.el_schema = yaml.safe_load(f) - with open(schema_path / "base_monte_carlo_aep_schema.yml") as f: + with (schema_path / "base_monte_carlo_aep_schema.yml").open() as f: self.mc_aep_schema = yaml.safe_load(f) - with open(schema_path / "base_tie_schema.yml") as f: + with (schema_path / "base_tie_schema.yml").open() as f: self.tie_schema = yaml.safe_load(f) - with open(schema_path / "scada_wake_losses_schema.yml") as f: + with (schema_path / "scada_wake_losses_schema.yml").open() as f: self.wake_schema = yaml.safe_load(f) def test_full_schema(self): diff --git a/test/regression/turbine_long_term_gross_energy.py b/test/regression/turbine_long_term_gross_energy.py index 3a0e4fc59..f6ed14757 100644 --- a/test/regression/turbine_long_term_gross_energy.py +++ b/test/regression/turbine_long_term_gross_energy.py @@ -6,6 +6,7 @@ from openoa.analysis import TurbineLongTermGrossEnergy + from test.conftest import project_ENGIE, example_data_path_str # isort: skip diff --git a/test/regression/wake_losses.py b/test/regression/wake_losses.py index b46849cdd..eaba57f79 100644 --- a/test/regression/wake_losses.py +++ b/test/regression/wake_losses.py @@ -2,12 +2,11 @@ import unittest import numpy as np -import pandas as pd -import pytest from numpy import testing as nptest from openoa.analysis import wake_losses + from test.conftest import project_ENGIE, example_data_path_str # isort: skip @@ -18,8 +17,7 @@ def reset_prng(): class TestWakeLosses(unittest.TestCase): def setUp(self): - """ - Python Unittest setUp method. + """Python Unittest setUp method. Load data from disk into PlantData objects and prepare the data for testing the WakeLosses method. """ reset_prng() diff --git a/test/regression/yaw_misalignment.py b/test/regression/yaw_misalignment.py index f28fecf0c..c79ba5c54 100644 --- a/test/regression/yaw_misalignment.py +++ b/test/regression/yaw_misalignment.py @@ -2,12 +2,12 @@ import unittest import numpy as np -import pandas as pd import pytest from numpy import testing as nptest from openoa.analysis import yaw_misalignment + from test.conftest import project_ENGIE, example_data_path_str # isort: skip @@ -18,8 +18,7 @@ def reset_prng(): class TestStaticYawMisalignment(unittest.TestCase): def setUp(self): - """ - Python Unittest setUp method. + """Python Unittest setUp method. Load data from disk into PlantData objects and prepare the data for testing the StaticYawMisalignment method. """ diff --git a/test/unit/test_converters.py b/test/unit/test_converters.py index 666b490bd..4c961381b 100644 --- a/test/unit/test_converters.py +++ b/test/unit/test_converters.py @@ -3,7 +3,6 @@ import numpy as np import pandas as pd import pytest -from numpy import testing as nptest from pandas import testing as tm from openoa.utils._converters import ( @@ -16,6 +15,7 @@ multiple_df_to_single_df, ) + test_df1 = pd.DataFrame( np.arange(15).reshape(5, 3, order="F"), columns=["a", "b", "c"], dtype=float ) @@ -58,7 +58,6 @@ def sample_df_handling_method( def test_list_of_len(): """Tests the `_list_of_len` method.""" - # Test for a 1 element list x = [1] length = 4 @@ -83,7 +82,6 @@ def test_list_of_len(): def test_convert_args_to_lists(): """Tests the `convert_args_to_lists` method.""" - # Test for a list that already contains lists x = [["a"], [5, 6]] length = 2 @@ -108,27 +106,26 @@ def test_convert_args_to_lists(): def test_df_to_series(): """Tests the `df_to_series` method.""" - # Test that each column is returned correctly, in a few different order variations y = [test_series_a1, test_series_b1, test_series_c1] y_test = df_to_series(test_df1, "a", "b", "c") - for el, el_test in zip(y, y_test): + for el, el_test in zip(y, y_test, strict=True): tm.assert_series_equal(el, el_test) y = [test_series_c1, test_series_a1] y_test = df_to_series(test_df1, "c", "a") - for el, el_test in zip(y, y_test): + for el, el_test in zip(y, y_test, strict=True): tm.assert_series_equal(el, el_test) y = [test_series_b1] y_test = df_to_series(test_df1, "b") - for el, el_test in zip(y, y_test): + for el, el_test in zip(y, y_test, strict=True): tm.assert_series_equal(el, el_test) # Test that None is returned for a passed None value y = [None, test_series_a1, None, test_series_b1, test_series_c1, None] y_test = df_to_series(test_df1, None, "a", None, "b", "c", None) - for el, el_test in zip(y, y_test): + for el, el_test in zip(y, y_test, strict=True): if el is None: assert el_test is None continue @@ -152,7 +149,6 @@ def test_df_to_series(): def test_multiple_df_to_single_df(): """Tests the `multiple_df_to_single_df` method.""" - # Test a basic working case with single column DataFrames y = test_df1 y_test = multiple_df_to_single_df(*test_df_list1) @@ -181,7 +177,6 @@ def test_multiple_df_to_single_df(): def test_series_to_df(): """Tests the `series_to_df` method.""" - # Test simple use case y = test_df1 y_test, (y_test_a1, y_test_b1, y_test_c1) = series_to_df( @@ -207,7 +202,7 @@ def test_series_to_df(): series_to_df(test_series_a1, test_series_b1, "c") # Ensure one input series maps correctly - for x, y in zip([test_series_a1, test_series_b1, test_series_c1], test_df_list1): + for x, y in zip([test_series_a1, test_series_b1, test_series_c1], test_df_list1, strict=True): y_test, [name] = series_to_df(x) tm.assert_frame_equal(y, y_test) assert name == x.name @@ -215,7 +210,6 @@ def test_series_to_df(): def test_series_method(): """Tests the `series_method` wrapper via `sample_series_handling_method()`.""" - # Ensure that the wrapper converts the string column names to series objects and the data kwarg to None y_test_a, y_test_c, y_test_df = sample_series_handling_method("a", 1.0, 2.0, "c", data=test_df1) tm.assert_series_equal(test_series_a1, y_test_a) @@ -240,7 +234,6 @@ def test_series_method(): def test_dataframe_method(): """Tests the `series_method` wrapper via `sample_df_handling_method()`.""" - # Ensure that the wrapper converts the Series to column names and the data to a DataFrame y = test_df1[["c", "a"]] y_test_c, y_test_a, y_test_df = sample_df_handling_method( diff --git a/test/unit/test_imputing_toolkit.py b/test/unit/test_imputing_toolkit.py index 63be87797..97d22740f 100644 --- a/test/unit/test_imputing_toolkit.py +++ b/test/unit/test_imputing_toolkit.py @@ -1,5 +1,4 @@ import unittest -from multiprocessing.sharedctypes import Value import numpy as np import pandas as pd diff --git a/test/unit/test_ml_toolkit.py b/test/unit/test_ml_toolkit.py index aa56275c0..96015effd 100644 --- a/test/unit/test_ml_toolkit.py +++ b/test/unit/test_ml_toolkit.py @@ -1,4 +1,3 @@ -import sys import unittest import numpy as np @@ -38,7 +37,7 @@ def test_algorithms(self): } # Loop through algorithms - for a in required_metrics.keys(): + for a in required_metrics: ml = MachineLearningSetup(a) # Setup ML object # Perform randomized grid search only once for efficiency diff --git a/test/unit/test_plant_helpers.py b/test/unit/test_plant_helpers.py index 2adf30447..c5492ea35 100644 --- a/test/unit/test_plant_helpers.py +++ b/test/unit/test_plant_helpers.py @@ -10,19 +10,10 @@ import pytest from attrs import field, define -from openoa.plant import ( # , compose_error_message - PlantData, - load_to_pandas, - rename_columns, - convert_to_list, - dtype_converter, - column_validator, - frequency_validator, - load_to_pandas_dict, -) -from openoa.schema import ( # , compose_error_message - ANALYSIS_REQUIREMENTS, - AssetMetaData, +from openoa.plant import rename_columns # , compose_error_message +from openoa.plant import convert_to_list, dtype_converter, column_validator, frequency_validator +from openoa.schema import AssetMetaData # , compose_error_message +from openoa.schema import ( FromDictMixin, MeterMetaData, PlantMetaData, @@ -39,6 +30,7 @@ deprecated_offset_map, ) + EXAMPLE_DATA_PATH = Path(__file__).resolve().parents[2] / "examples/data" @@ -52,13 +44,13 @@ def test_convert_frequency(): with pytest.warns(DeprecationWarning): convert_frequency(offset) - assert "ME" == convert_frequency("M") - assert "1h" == convert_frequency("1H") - assert "10min" == convert_frequency("10T") - assert "20s" == convert_frequency("20S") - assert "ms" == convert_frequency("L") - assert "us" == convert_frequency("U") - assert "ns" == convert_frequency("N") + assert convert_frequency("M") == "ME" + assert convert_frequency("1H") == "1h" + assert convert_frequency("10T") == "10min" + assert convert_frequency("20S") == "20s" + assert convert_frequency("L") == "ms" + assert convert_frequency("U") == "us" + assert convert_frequency("N") == "ns" with pytest.raises(ValueError): convert_frequency("10min1") @@ -116,7 +108,6 @@ def test_frequency_validator() -> None: """Tests the `frequency_validator` method. All inputs are formatted to the desired input, so testing of the input types is not required. """ - # Test None as desired frequency returns True always assert frequency_validator("anything", None, exact=True) assert frequency_validator("anything", None, exact=False) @@ -133,9 +124,9 @@ def test_frequency_validator() -> None: desired_valid_2 = ("10min", "h", "ns") # set of options case desired_invalid = _at_least_hourly # set of non exact matches - assert frequency_validator(actual, desired_valid_1, True) - assert frequency_validator(actual, desired_valid_2, True) - assert not frequency_validator(actual, desired_invalid, True) + assert frequency_validator(actual, desired_valid_1, exact=True) + assert frequency_validator(actual, desired_valid_2, exact=True) + assert not frequency_validator(actual, desired_invalid, exact=True) # Test for non-exact matches actual_1 = "10min" @@ -150,19 +141,18 @@ def test_frequency_validator() -> None: "h", ) # set of greater than or equal to hourly frequency resolutions - assert frequency_validator(actual_1, desired_valid, False) - assert frequency_validator(actual_2, desired_valid, False) - assert frequency_validator(actual_3, desired_valid, False) - assert not frequency_validator(actual_1, desired_invalid, False) - assert not frequency_validator(actual_2, desired_invalid, False) - assert not frequency_validator(actual_3, desired_invalid, False) + assert frequency_validator(actual_1, desired_valid, exact=False) + assert frequency_validator(actual_2, desired_valid, exact=False) + assert frequency_validator(actual_3, desired_valid, exact=False) + assert not frequency_validator(actual_1, desired_invalid, exact=False) + assert not frequency_validator(actual_2, desired_invalid, exact=False) + assert not frequency_validator(actual_3, desired_invalid, exact=False) def test_convert_to_list(): """Tests the converter function for turning single inputs into a list of input, - or applying a manipulation across a list of inputs + or applying a manipulation across a list of inputs. """ - # Test that a list of the value is returned assert convert_to_list(1) == [1] assert convert_to_list(None) == [None] @@ -220,13 +210,24 @@ def test_dtype_converter(): df.string_col = np.arange(7) df.problem_col = ["one", "two", "string", "invalid", 5, 6.0, 7] - column_types_invalid_1 = dict( - time=pd.DatetimeIndex, float_col=float, string_col=str, problem_col=float - ) - column_types_invalid_2 = dict( - time=np.datetime64, float_col=float, string_col=str, problem_col=int - ) - column_types_valid = dict(time=np.datetime64, float_col=float, string_col=str, problem_col=str) + column_types_invalid_1 = { + "time": pd.DatetimeIndex, + "float_col": float, + "string_col": str, + "problem_col": float, + } + column_types_invalid_2 = { + "time": np.datetime64, + "float_col": float, + "string_col": str, + "problem_col": int, + } + column_types_valid = { + "time": np.datetime64, + "float_col": float, + "string_col": str, + "problem_col": str, + } assert dtype_converter(df, column_types_invalid_1) == ["problem_col"] assert dtype_converter(df, column_types_invalid_2) == ["problem_col"] @@ -274,18 +275,18 @@ def test_SCADAMetaData(): # Tests the SCADAMetaData for defaults and user-provided values # Leaving asset_id and power as the default values - meta_dict = dict( - time="datetime", - WMET_HorWdSpd="ws_100", - WMET_HorWdDir="wd_100", - WMET_HorWdDirRel="wd_rel_100", - WTUR_TurSt="turb_stat", - WROT_BlPthAngVal="rotor_angle", - WMET_EnvTmp="temp", - frequency="h", - ) + meta_dict = { + "time": "datetime", + "WMET_HorWdSpd": "ws_100", + "WMET_HorWdDir": "wd_100", + "WMET_HorWdDirRel": "wd_rel_100", + "WTUR_TurSt": "turb_stat", + "WROT_BlPthAngVal": "rotor_angle", + "WMET_EnvTmp": "temp", + "frequency": "h", + } valid_map = deepcopy(meta_dict) - valid_map.update(dict(asset_id="asset_id", WTUR_W="WTUR_W")) + valid_map.update({"asset_id": "asset_id", "WTUR_W": "WTUR_W"}) valid_map.pop("frequency") meta = SCADAMetaData.from_dict(meta_dict) @@ -310,11 +311,9 @@ def test_MeterMetaData(): # Tests the MeterMetaData for defaults and user-provided values # Leaving time and energy as the default values - meta_dict = dict( - frequency="D", - ) + meta_dict = {"frequency": "D"} valid_map = deepcopy(meta_dict) - valid_map.update(dict(time="time", MMTR_SupWh="MMTR_SupWh")) + valid_map.update({"time": "time", "MMTR_SupWh": "MMTR_SupWh"}) valid_map.pop("frequency") meta = MeterMetaData.from_dict(meta_dict) @@ -337,15 +336,15 @@ def test_TowerMetaData(): # Tests the TowerMetaData for defaults and user-provided values # Leaving time as the default value - meta_dict = dict( - asset_id="the_IDs", - frequency="D", - WMET_HorWdSpd="windspeed", - WMET_HorWdDir="winddir", - WMET_EnvTmp="TempC", - ) + meta_dict = { + "asset_id": "the_IDs", + "frequency": "D", + "WMET_HorWdSpd": "windspeed", + "WMET_HorWdDir": "winddir", + "WMET_EnvTmp": "TempC", + } valid_map = deepcopy(meta_dict) - valid_map.update(dict(time="time")) + valid_map.update({"time": "time"}) valid_map.pop("frequency") meta = TowerMetaData.from_dict(meta_dict) @@ -368,14 +367,14 @@ def test_StatusMetaData(): # Tests the StatusMetaData for defaults and user-provided values # Leaving time and status_text as the default values - meta_dict = dict( - asset_id="the_IDs", - status_id="status_ids", - status_code="code", - frequency="h", - ) + meta_dict = { + "asset_id": "the_IDs", + "status_id": "status_ids", + "status_code": "code", + "frequency": "h", + } valid_map = deepcopy(meta_dict) - valid_map.update(dict(time="time", status_text="status_text")) + valid_map.update({"time": "time", "status_text": "status_text"}) valid_map.pop("frequency") meta = StatusMetaData.from_dict(meta_dict) @@ -398,13 +397,13 @@ def test_CurtailMetaData(): # Tests the CurtailMetaData for defaults and user-provided values # Leaving time and net_energy as the default values - meta_dict = dict( - IAVL_ExtPwrDnWh="curtail", - IAVL_DnWh="avail", - frequency="h", - ) + meta_dict = { + "IAVL_ExtPwrDnWh": "curtail", + "IAVL_DnWh": "avail", + "frequency": "h", + } valid_map = deepcopy(meta_dict) - valid_map.update(dict(time="time")) + valid_map.update({"time": "time"}) valid_map.pop("frequency") meta = CurtailMetaData.from_dict(meta_dict) @@ -427,16 +426,16 @@ def test_AssetMetaData(): # Tests the AssetMetaData for defaults and user-provided values # Leaving elevation and type as the default values - meta_dict = dict( - asset_id="asset_name", - latitude="lat", - longitude="lon", - rated_power="P", - hub_height="HH", - rotor_diameter="RD", - ) + meta_dict = { + "asset_id": "asset_name", + "latitude": "lat", + "longitude": "lon", + "rated_power": "P", + "hub_height": "HH", + "rotor_diameter": "RD", + } valid_map = deepcopy(meta_dict) - valid_map.update(dict(elevation="elevation", type="type")) + valid_map.update({"elevation": "elevation", "type": "type"}) meta = AssetMetaData.from_dict(meta_dict) assert meta.col_map == valid_map @@ -457,16 +456,16 @@ def test_ReanalysisMetaData(): # Tests the ReanalysisMetaData for defaults and user-provided values # Leaving temperature, density, and frequency as the default values - meta_dict = dict( - time="curtail", - WMETR_HorWdSpd="WS", - WMETR_HorWdSpdU="ws_U", - WMETR_HorWdSpdV="ws_V", - WMETR_HorWdDir="wdir", - WMETR_EnvPres="pressure", - ) + meta_dict = { + "time": "curtail", + "WMETR_HorWdSpd": "WS", + "WMETR_HorWdSpdU": "ws_U", + "WMETR_HorWdSpdV": "ws_V", + "WMETR_HorWdDir": "wdir", + "WMETR_EnvPres": "pressure", + } valid_map = deepcopy(meta_dict) - valid_map.update(dict(WMETR_EnvTmp="WMETR_EnvTmp", WMETR_AirDen="WMETR_AirDen")) + valid_map.update({"WMETR_EnvTmp": "WMETR_EnvTmp", "WMETR_AirDen": "WMETR_AirDen"}) meta = ReanalysisMetaData.from_dict(meta_dict) assert meta.col_map == valid_map @@ -488,33 +487,33 @@ def test_convert_reanalysis_value(): # Test the ReanalysisMetaData dictionary converter method # Leaving the merra2 key as all defaults - era5_meta_dict = dict( - time="curtail", - WMETR_HorWdSpd="WS", - WMETR_HorWdSpdU="ws_U", - WMETR_HorWdSpdV="ws_V", - WMETR_HorWdDir="wdir", - WMETR_EnvTmp="temps", - WMETR_AirDen="dens", - WMETR_EnvPres="pressure", - frequency="5min", - ) + era5_meta_dict = { + "time": "curtail", + "WMETR_HorWdSpd": "WS", + "WMETR_HorWdSpdU": "ws_U", + "WMETR_HorWdSpdV": "ws_V", + "WMETR_HorWdDir": "wdir", + "WMETR_EnvTmp": "temps", + "WMETR_AirDen": "dens", + "WMETR_EnvPres": "pressure", + "frequency": "5min", + } valid_era5_map = deepcopy(era5_meta_dict) valid_era5_map.pop("frequency") # Copy of the defaults - valid_merra2_map = dict( - time="time", - WMETR_HorWdSpd="windspeed", - WMETR_HorWdSpdU="windspeed_u", - WMETR_HorWdSpdV="windspeed_v", - WMETR_HorWdDir="wind_direction", - WMETR_EnvTmp="temperature", - WMETR_AirDen="density", - WMETR_EnvPres="surface_pressure", - ) + valid_merra2_map = { + "time": "time", + "WMETR_HorWdSpd": "windspeed", + "WMETR_HorWdSpdU": "windspeed_u", + "WMETR_HorWdSpdV": "windspeed_v", + "WMETR_HorWdDir": "wind_direction", + "WMETR_EnvTmp": "temperature", + "WMETR_AirDen": "density", + "WMETR_EnvPres": "surface_pressure", + } - meta = convert_reanalysis(value=dict(era5=era5_meta_dict, merra2=dict())) + meta = convert_reanalysis(value={"era5": era5_meta_dict, "merra2": {}}) assert meta["era5"].col_map == valid_era5_map assert meta["era5"].frequency == era5_meta_dict["frequency"] @@ -522,7 +521,7 @@ def test_convert_reanalysis_value(): assert meta["era5"].units == attr.fields(ReanalysisMetaData).units.default assert meta["era5"].dtypes == attr.fields(ReanalysisMetaData).dtypes.default - meta = convert_reanalysis(value=dict(era5=dict(), merra2=valid_merra2_map)) + meta = convert_reanalysis(value={"era5": {}, "merra2": valid_merra2_map}) assert meta["merra2"].col_map == valid_merra2_map assert meta["merra2"].frequency == attr.fields(ReanalysisMetaData).frequency.default assert meta["merra2"].units == attr.fields(ReanalysisMetaData).units.default @@ -558,8 +557,8 @@ def test_PlantMetaData_defaults(): assert vals["reanalysis"] == {"product": ReanalysisMetaData().col_map} # Check the defaults for an empty reanalysis input - meta = PlantMetaData(reanalysis=dict(era5=ReanalysisMetaData().col_map)) - assert meta.reanalysis == dict(era5=ReanalysisMetaData()) + meta = PlantMetaData(reanalysis={"era5": ReanalysisMetaData().col_map}) + assert meta.reanalysis == {"era5": ReanalysisMetaData()} vals = meta.column_map assert vals["reanalysis"]["era5"] == ReanalysisMetaData().col_map diff --git a/test/unit/test_power_curve_toolkit.py b/test/unit/test_power_curve_toolkit.py index bcf7544b9..ef0f226bc 100644 --- a/test/unit/test_power_curve_toolkit.py +++ b/test/unit/test_power_curve_toolkit.py @@ -7,6 +7,7 @@ from openoa.utils import power_curve from openoa.utils.power_curve.parametric_forms import logistic5param, logistic5param_capped + noise = 0.1 diff --git a/test/unit/test_timeseries_toolkit.py b/test/unit/test_timeseries_toolkit.py index 831903de2..5f5f27253 100644 --- a/test/unit/test_timeseries_toolkit.py +++ b/test/unit/test_timeseries_toolkit.py @@ -30,7 +30,7 @@ def setUp(self): def test_convert_local_to_utc(self): # Pass in a localized datetime with matching tz string and make sure it throws an exception - self.assertRaises( + self.assertRaises( # noqa: B017 Exception, self.mountain_tz.localize(self.summer_midnight), "T1: No exception raised for a datetime object with baked in TZInfo", @@ -156,7 +156,7 @@ def test_percent_nan(self): nan_values = {"a": 0.0, "b": 0.2, "c": 0.4} - for a, b in test_dict.items(): + for a in test_dict: nptest.assert_almost_equal( nan_values[a], timeseries.percent_nan(test_dict[a]),