diff --git a/config/config.yaml b/config/config.yaml index 5827451..874a9d9 100644 --- a/config/config.yaml +++ b/config/config.yaml @@ -118,7 +118,7 @@ imputation: utility pv: "land" time: scenario: "construction" # Options: historical->construction->pre_construction->announced - start_year_imputation_method: capacity_profile + method: capacity_profile lifetime_years: # General lifetime assumptions. Adjust as needed! bioenergy turbine: 25 diff --git a/pixi.lock b/pixi.lock index 835b1e6..9222d6e 100644 --- a/pixi.lock +++ b/pixi.lock @@ -1,8 +1,20 @@ version: 7 platforms: - name: linux-64 + virtual-packages: + - __unix=0=0 + - __linux=4.18 + - __glibc=2.28 + - __archspec=0=x86_64 - name: osx-arm64 + virtual-packages: + - __unix=0=0 + - __osx=13.0 + - __archspec=0=m1 - name: win-64 + virtual-packages: + - __win=10.0 + - __archspec=0=x86_64 environments: default: channels: @@ -1053,7 +1065,7 @@ environments: - conda: https://conda.anaconda.org/conda-forge/linux-64/openpyxl-3.1.5-py312h7f6eeab_3.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/openssl-3.6.3-h35e630c_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/orc-2.2.2-hbb90d81_1.conda - - conda: https://conda.anaconda.org/conda-forge/linux-64/pandas-2.3.0-py312hf9745cd_0.conda + - conda: https://conda.anaconda.org/conda-forge/linux-64/pandas-3.0.5-py312h8ecdadd_1.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/pcre2-10.46-h1321c63_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/pillow-12.2.0-py312h50c33e8_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/pixman-0.46.4-h54a6638_1.conda @@ -1142,8 +1154,8 @@ environments: - conda: https://conda.anaconda.org/conda-forge/noarch/fonts-conda-ecosystem-1-0.tar.bz2 - conda: https://conda.anaconda.org/conda-forge/noarch/fonts-conda-forge-1-hc364b38_1.conda - conda: https://conda.anaconda.org/conda-forge/noarch/fsspec-2026.6.0-pyhd8ed1ab_0.conda - - conda: https://conda.anaconda.org/conda-forge/noarch/geopandas-1.0.1-pyhd8ed1ab_3.conda - - conda: https://conda.anaconda.org/conda-forge/noarch/geopandas-base-1.0.1-pyha770c72_3.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/geopandas-1.1.4-pyhd8ed1ab_0.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/geopandas-base-1.1.4-pyha770c72_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/gregor-0.1.0-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/h2-4.3.0-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/hpack-4.2.0-pyhd8ed1ab_0.conda @@ -1164,9 +1176,9 @@ environments: - conda: https://conda.anaconda.org/conda-forge/noarch/narwhals-2.22.1-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/networkx-3.6.1-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/packaging-26.2-pyhc364b38_0.conda - - conda: https://conda.anaconda.org/conda-forge/noarch/pandera-0.24.0-hd8ed1ab_2.conda - - conda: https://conda.anaconda.org/conda-forge/noarch/pandera-base-0.24.0-pyhd8ed1ab_2.conda - - conda: https://conda.anaconda.org/conda-forge/noarch/pandera-geopandas-0.24.0-hd8ed1ab_2.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/pandera-0.32.1-ha00cc4c_0.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/pandera-base-0.32.1-pyhcf101f3_0.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/pandera-geopandas-0.32.1-h9ad522b_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/parso-0.8.7-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/partd-1.4.2-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/pexpect-4.9.0-pyhd8ed1ab_1.conda @@ -1178,9 +1190,7 @@ environments: - conda: https://conda.anaconda.org/conda-forge/noarch/pyparsing-3.3.2-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/pysocks-1.7.1-pyha55dd90_7.conda - conda: https://conda.anaconda.org/conda-forge/noarch/python-dateutil-2.9.0.post0-pyhe01879c_2.conda - - conda: https://conda.anaconda.org/conda-forge/noarch/python-tzdata-2026.2-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/python_abi-3.12-8_cp312.conda - - conda: https://conda.anaconda.org/conda-forge/noarch/pytz-2026.2-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/rasterstats-0.20.0-pyhd8ed1ab_2.conda - conda: https://conda.anaconda.org/conda-forge/noarch/requests-2.34.2-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/rioxarray-0.22.0-pyhc364b38_0.conda @@ -1231,8 +1241,8 @@ environments: - conda: https://conda.anaconda.org/conda-forge/noarch/executing-2.2.1-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/folium-0.20.0-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/fsspec-2026.6.0-pyhd8ed1ab_0.conda - - conda: https://conda.anaconda.org/conda-forge/noarch/geopandas-1.0.1-pyhd8ed1ab_3.conda - - conda: https://conda.anaconda.org/conda-forge/noarch/geopandas-base-1.0.1-pyha770c72_3.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/geopandas-1.1.4-pyhd8ed1ab_0.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/geopandas-base-1.1.4-pyha770c72_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/gregor-0.1.0-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/h2-4.3.0-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/hpack-4.2.0-pyhd8ed1ab_0.conda @@ -1253,9 +1263,9 @@ environments: - conda: https://conda.anaconda.org/conda-forge/noarch/narwhals-2.22.1-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/networkx-3.6.1-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/packaging-26.2-pyhc364b38_0.conda - - conda: https://conda.anaconda.org/conda-forge/noarch/pandera-0.24.0-hd8ed1ab_2.conda - - conda: https://conda.anaconda.org/conda-forge/noarch/pandera-base-0.24.0-pyhd8ed1ab_2.conda - - conda: https://conda.anaconda.org/conda-forge/noarch/pandera-geopandas-0.24.0-hd8ed1ab_2.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/pandera-0.32.1-ha00cc4c_0.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/pandera-base-0.32.1-pyhcf101f3_0.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/pandera-geopandas-0.32.1-h9ad522b_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/parso-0.8.7-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/partd-1.4.2-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/pexpect-4.9.0-pyhd8ed1ab_1.conda @@ -1267,9 +1277,7 @@ environments: - conda: https://conda.anaconda.org/conda-forge/noarch/pyparsing-3.3.2-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/pysocks-1.7.1-pyha55dd90_7.conda - conda: https://conda.anaconda.org/conda-forge/noarch/python-dateutil-2.9.0.post0-pyhe01879c_2.conda - - conda: https://conda.anaconda.org/conda-forge/noarch/python-tzdata-2026.2-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/python_abi-3.12-8_cp312.conda - - conda: https://conda.anaconda.org/conda-forge/noarch/pytz-2026.2-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/rasterstats-0.20.0-pyhd8ed1ab_2.conda - conda: https://conda.anaconda.org/conda-forge/noarch/requests-2.34.2-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/rioxarray-0.22.0-pyhc364b38_0.conda @@ -1411,7 +1419,7 @@ environments: - conda: https://conda.anaconda.org/conda-forge/osx-arm64/openpyxl-3.1.5-py312h2a925e6_3.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/openssl-3.6.3-hd24854e_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/orc-2.2.2-h578b684_1.conda - - conda: https://conda.anaconda.org/conda-forge/osx-arm64/pandas-2.3.0-py312hcb1e3ce_0.conda + - conda: https://conda.anaconda.org/conda-forge/osx-arm64/pandas-3.0.5-py312h6510ced_1.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/pcre2-10.46-h7125dd6_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/pillow-12.2.0-py312h4e908a4_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/proj-9.6.2-hdbeaa80_2.conda @@ -1477,8 +1485,8 @@ environments: - conda: https://conda.anaconda.org/conda-forge/noarch/fonts-conda-ecosystem-1-0.tar.bz2 - conda: https://conda.anaconda.org/conda-forge/noarch/fonts-conda-forge-1-hc364b38_1.conda - conda: https://conda.anaconda.org/conda-forge/noarch/fsspec-2026.6.0-pyhd8ed1ab_0.conda - - conda: https://conda.anaconda.org/conda-forge/noarch/geopandas-1.0.1-pyhd8ed1ab_3.conda - - conda: https://conda.anaconda.org/conda-forge/noarch/geopandas-base-1.0.1-pyha770c72_3.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/geopandas-1.1.4-pyhd8ed1ab_0.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/geopandas-base-1.1.4-pyha770c72_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/gregor-0.1.0-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/h2-4.3.0-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/hpack-4.2.0-pyhd8ed1ab_0.conda @@ -1499,9 +1507,9 @@ environments: - conda: https://conda.anaconda.org/conda-forge/noarch/narwhals-2.22.1-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/networkx-3.6.1-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/packaging-26.2-pyhc364b38_0.conda - - conda: https://conda.anaconda.org/conda-forge/noarch/pandera-0.24.0-hd8ed1ab_2.conda - - conda: https://conda.anaconda.org/conda-forge/noarch/pandera-base-0.24.0-pyhd8ed1ab_2.conda - - conda: https://conda.anaconda.org/conda-forge/noarch/pandera-geopandas-0.24.0-hd8ed1ab_2.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/pandera-0.32.1-ha00cc4c_0.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/pandera-base-0.32.1-pyhcf101f3_0.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/pandera-geopandas-0.32.1-h9ad522b_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/parso-0.8.7-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/partd-1.4.2-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/prompt-toolkit-3.0.52-pyha770c72_0.conda @@ -1513,7 +1521,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/noarch/python-dateutil-2.9.0.post0-pyhe01879c_2.conda - conda: https://conda.anaconda.org/conda-forge/noarch/python-tzdata-2026.2-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/python_abi-3.12-8_cp312.conda - - conda: https://conda.anaconda.org/conda-forge/noarch/pytz-2026.2-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/rasterstats-0.20.0-pyhd8ed1ab_2.conda - conda: https://conda.anaconda.org/conda-forge/noarch/requests-2.34.2-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/rioxarray-0.22.0-pyhc364b38_0.conda @@ -1651,7 +1658,7 @@ environments: - conda: https://conda.anaconda.org/conda-forge/win-64/openpyxl-3.1.5-py312h83acffa_3.conda - conda: https://conda.anaconda.org/conda-forge/win-64/openssl-3.6.3-hf411b9b_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/orc-2.2.2-h0a1ad0e_1.conda - - conda: https://conda.anaconda.org/conda-forge/win-64/pandas-2.3.0-py312h72972c8_0.conda + - conda: https://conda.anaconda.org/conda-forge/win-64/pandas-3.0.5-py312h95189c4_1.conda - conda: https://conda.anaconda.org/conda-forge/win-64/pcre2-10.46-h3402e2f_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/pillow-12.2.0-py312h31f0997_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/pixman-0.46.4-h5112557_1.conda @@ -5932,57 +5939,6 @@ packages: - orc >=2.3.0,<2.3.1.0a0 size: 1468651 timestamp: 1773230208923 -- conda: https://conda.anaconda.org/conda-forge/linux-64/pandas-2.3.0-py312hf9745cd_0.conda - sha256: 44f5587c1e1a9f0257387dd18735bcf65a67a6089e723302dc7947be09d9affe - md5: ac82ac336dbe61106e21fb2e11704459 - depends: - - __glibc >=2.17,<3.0.a0 - - libgcc >=13 - - libstdcxx >=13 - - numpy >=1.19,<3 - - numpy >=1.22.4 - - python >=3.12,<3.13.0a0 - - python-dateutil >=2.8.2 - - python-tzdata >=2022.7 - - python_abi 3.12.* *_cp312 - - pytz >=2020.1 - constrains: - - bottleneck >=1.3.6 - - blosc >=1.21.3 - - numba >=0.56.4 - - pyqt5 >=5.15.9 - - pyarrow >=10.0.1 - - gcsfs >=2022.11.0 - - xlsxwriter >=3.0.5 - - scipy >=1.10.0 - - beautifulsoup4 >=4.11.2 - - numexpr >=2.8.4 - - fastparquet >=2022.12.0 - - lxml >=4.9.2 - - xlrd >=2.0.1 - - openpyxl >=3.1.0 - - qtpy >=2.3.0 - - s3fs >=2022.11.0 - - pandas-gbq >=0.19.0 - - pytables >=3.8.0 - - python-calamine >=0.1.7 - - fsspec >=2022.11.0 - - psycopg2 >=2.9.6 - - xarray >=2022.12.0 - - matplotlib >=3.6.3 - - pyxlsb >=1.0.10 - - tabulate >=0.9.0 - - odfpy >=1.4.1 - - pyreadstat >=1.2.0 - - html5lib >=1.1 - - zstandard >=0.19.0 - - sqlalchemy >=2.0.0 - - tzdata >=2022.7 - license: BSD-3-Clause - license_family: BSD - run_exports: {} - size: 14958450 - timestamp: 1749100123120 - conda: https://conda.anaconda.org/conda-forge/linux-64/pandas-2.3.1-py312hf79963d_0.conda sha256: 6ec86b1da8432059707114270b9a45d767dac97c4910ba82b1f4fa6f74e077c8 md5: 7c73e62e62e5864b8418440e2a2cc246 @@ -6035,6 +5991,62 @@ packages: - pkg:pypi/pandas?source=hash-mapping size: 15092371 timestamp: 1752082221274 +- conda: https://conda.anaconda.org/conda-forge/linux-64/pandas-3.0.5-py312h8ecdadd_1.conda + sha256: 393e529c0574c020a9b790275af10ff35e21cbb4e12740888bd3f214f2f07034 + md5: 85eb29ade84d1bd97be5ec55607efb00 + depends: + - python + - numpy >=1.26.0 + - python-dateutil >=2.8.2 + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - numpy >=1.23,<3 + - python_abi 3.12.* *_cp312 + constrains: + - adbc-driver-postgresql >=1.2.0 + - adbc-driver-sqlite >=1.2.0 + - beautifulsoup4 >=4.12.3 + - blosc >=1.21.3 + - bottleneck >=1.4.2 + - fastparquet >=2024.11.0 + - fsspec >=2024.10.0 + - gcsfs >=2024.10.0 + - html5lib >=1.1 + - hypothesis >=6.116.0 + - jinja2 >=3.1.5 + - lxml >=5.3.0 + - matplotlib >=3.9.3 + - numba >=0.60.0 + - numexpr >=2.10.2 + - odfpy >=1.4.1 + - openpyxl >=3.1.5 + - psycopg2 >=2.9.10 + - pyarrow >=13.0.0 + - pyiceberg >=0.8.1 + - pymysql >=1.1.1 + - pyqt5 >=5.15.9 + - pyreadstat >=1.2.8 + - pytables >=3.10.1 + - pytest >=8.3.4 + - pytest-xdist >=3.6.1 + - python-calamine >=0.3.0 + - pytz >=2024.2 + - pyxlsb >=1.0.10 + - qtpy >=2.4.2 + - scipy >=1.14.1 + - s3fs >=2024.10.0 + - sqlalchemy >=2.0.36 + - tabulate >=0.9.0 + - xarray >=2024.10.0 + - xlrd >=2.0.1 + - xlsxwriter >=3.2.0 + - zstandard >=0.23.0 + license: BSD-3-Clause + license_family: BSD + run_exports: {} + size: 14936796 + timestamp: 1785276227779 - conda: https://conda.anaconda.org/conda-forge/linux-64/pango-1.56.4-hadf4263_0.conda sha256: 3613774ad27e48503a3a6a9d72017087ea70f1426f6e5541dbdb59a3b626eaaf md5: 79f71230c069a287efe3a8614069ddf1 @@ -8328,37 +8340,37 @@ packages: run_exports: {} size: 149709 timestamp: 1781615868173 -- conda: https://conda.anaconda.org/conda-forge/noarch/geopandas-1.0.1-pyhd8ed1ab_3.conda - sha256: 04f7e616ebbf6352ff852b53c57901e43f14e2b3c92411f99b5547f106bc192e - md5: 1baca589eb35814a392eaad6d152447e +- conda: https://conda.anaconda.org/conda-forge/noarch/geopandas-1.1.4-pyhd8ed1ab_0.conda + sha256: cd478c0835106afce2478d6dcc525854219c4e710c7b446e2dd5042bfc9da082 + md5: 99d77e0f92f7b0b977f77cfb49c7eaf6 depends: - folium - - geopandas-base 1.0.1 pyha770c72_3 - - mapclassify >=2.4.0 + - geopandas-base 1.1.4 pyha770c72_0 + - mapclassify >=2.5.0 - matplotlib-base - pyogrio >=0.7.2 - - pyproj >=3.3.0 - - python >=3.9 + - pyproj >=3.5.0 + - python >=3.10 - xyzservices license: BSD-3-Clause license_family: BSD run_exports: {} - size: 7583 - timestamp: 1734346218849 -- conda: https://conda.anaconda.org/conda-forge/noarch/geopandas-base-1.0.1-pyha770c72_3.conda - sha256: 2d031871b57c6d4e5e2d6cc23bd6d4e0084bb52ebca5c1b20bf06d03749e0f24 - md5: e8343d1b635bf09dafdd362d7357f395 + size: 8267 + timestamp: 1782507446937 +- conda: https://conda.anaconda.org/conda-forge/noarch/geopandas-base-1.1.4-pyha770c72_0.conda + sha256: 77d0527d91c2a4bba135fdffb30557c88bacc22d7440c6632502609dd1ef0ab6 + md5: 6e31007c02f361338281bb78c5b84e7c depends: - - numpy >=1.22 + - numpy >=1.24 - packaging - - pandas >=1.4.0 - - python >=3.9 + - pandas >=2.0.0 + - python >=3.10 - shapely >=2.0.0 license: BSD-3-Clause license_family: BSD run_exports: {} - size: 239261 - timestamp: 1734346217454 + size: 255909 + timestamp: 1782507445933 - conda: https://conda.anaconda.org/conda-forge/noarch/gitdb-4.0.12-pyhd8ed1ab_0.conda sha256: dbbec21a369872c8ebe23cb9a3b9d63638479ee30face165aa0fccc96e93eec3 md5: 7c14f3706e099f8fcd47af2d494616cc @@ -9066,18 +9078,6 @@ packages: run_exports: {} size: 91574 timestamp: 1777103621679 -- conda: https://conda.anaconda.org/conda-forge/noarch/pandera-0.24.0-hd8ed1ab_2.conda - sha256: 857657552f1a8441e4c9e723637b8270f46b6262403dc18a76559dd5a5ba782e - md5: a12a21a89519f0cc224e44da86f5be2c - depends: - - numpy >=1.24.4 - - pandas >=2.1.1 - - pandera-base 0.24.0 pyhd8ed1ab_2 - license: MIT - license_family: MIT - run_exports: {} - size: 7364 - timestamp: 1749059959229 - conda: https://conda.anaconda.org/conda-forge/noarch/pandera-0.25.0-hd8ed1ab_1.conda sha256: b4eb7857d927b9001a2fdc11ee70add246c2de80e1e08bff3a8b67ce0cdc7912 md5: c9dca5dbec0de5c56e248087ba18ac02 @@ -9090,20 +9090,18 @@ packages: purls: [] size: 7458 timestamp: 1752079800481 -- conda: https://conda.anaconda.org/conda-forge/noarch/pandera-base-0.24.0-pyhd8ed1ab_2.conda - sha256: 98b3e59268d73824fabc3535fbe48f5973eb70f3975686bcf35949a9f3daaa17 - md5: e1c50e117a98e39d297d9290132f032b +- conda: https://conda.anaconda.org/conda-forge/noarch/pandera-0.32.1-ha00cc4c_0.conda + sha256: c686bea5de04fd6a93de31c9be18bfd24b9efe672a5e2e8692d34a35b282cbb8 + md5: f6b96a51cda3cf9a068fd748c8155ced depends: - - packaging >=20.0 - - pydantic - - python >=3.9 - - typeguard - - typing_inspect >=0.6.0 + - pandera-base ==0.32.1 pyhcf101f3_0 + - pandas >=2.1.1 + - numpy >=1.24.4 license: MIT license_family: MIT run_exports: {} - size: 154521 - timestamp: 1749059957954 + size: 4464 + timestamp: 1782916642380 - conda: https://conda.anaconda.org/conda-forge/noarch/pandera-base-0.25.0-pyhd8ed1ab_1.conda sha256: 98c3b93e690426dbdd5ef788db9b183bc75202ebbc563ed1859df39da2f86e8f md5: 8f88cb3ba3aac2992171892cd5f6d48d @@ -9119,18 +9117,34 @@ packages: - pkg:pypi/pandera?source=hash-mapping size: 164068 timestamp: 1752079799520 -- conda: https://conda.anaconda.org/conda-forge/noarch/pandera-geopandas-0.24.0-hd8ed1ab_2.conda - sha256: 53d1591315e0688c796af5d70b426536e704f9c8d3fdc2a650eb14a3ecfc7f88 - md5: fb21509f073465506ea994dc4667bf66 +- conda: https://conda.anaconda.org/conda-forge/noarch/pandera-base-0.32.1-pyhcf101f3_0.conda + sha256: 81eb4d5f49a2f98b7dbeb60b134ecef4b90cdf21b2af8a024d5a0fe2858e914b + md5: c78daae943cb94e6e6b2eab16e16a18f + depends: + - python >=3.10 + - packaging >=20.0 + - pydantic + - typeguard + - typing_extensions + - typing_inspect >=0.6.0 + - python + license: MIT + license_family: MIT + run_exports: {} + size: 268608 + timestamp: 1782916642380 +- conda: https://conda.anaconda.org/conda-forge/noarch/pandera-geopandas-0.32.1-h9ad522b_0.conda + sha256: 9c004523994f27e09ec39346bc4f57d9ac6cd5ca2b1c1e3beb8c9d7cf3e38c13 + md5: b6b27c3f219ce4d01c8ea31d435f263c depends: + - pandera ==0.32.1 ha00cc4c_0 - geopandas - - pandera 0.24.0 hd8ed1ab_2 - shapely license: MIT license_family: MIT run_exports: {} - size: 7429 - timestamp: 1749059961525 + size: 4646 + timestamp: 1782916642380 - conda: https://conda.anaconda.org/conda-forge/noarch/parso-0.8.4-pyhd8ed1ab_1.conda sha256: 17131120c10401a99205fc6fe436e7903c0fa092f1b3e80452927ab377239bcc md5: 5c092057b6badd30f75b06244ecd01c9 @@ -9533,17 +9547,6 @@ packages: - pkg:pypi/pytz?source=compressed-mapping size: 201725 timestamp: 1773679724369 -- conda: https://conda.anaconda.org/conda-forge/noarch/pytz-2026.2-pyhcf101f3_0.conda - sha256: 5020863d629f584b5c057333a67a7aed43e3ed013ba15dd70f353501ccb5aff6 - md5: 03cb60f505ad3ada0a95277af5faeb1a - depends: - - python >=3.10 - - python - license: MIT - license_family: MIT - run_exports: {} - size: 201747 - timestamp: 1777892201250 - conda: https://conda.anaconda.org/conda-forge/noarch/rasterstats-0.20.0-pyhd8ed1ab_2.conda sha256: 8ac6ad2fa2984842afa0e03d661328fdffb6965cd0a0eaeef2eadb561901068b md5: a6762dbfc4cf7084adf09f0accdf0f8a @@ -13467,57 +13470,6 @@ packages: - orc >=2.3.0,<2.3.1.0a0 size: 548180 timestamp: 1773230270828 -- conda: https://conda.anaconda.org/conda-forge/osx-arm64/pandas-2.3.0-py312hcb1e3ce_0.conda - sha256: 3105a94036f37429ed292763d3034008fd0b4911bd565bdf86c33e898655dcdf - md5: d95b29a40430115d6aa817f70be5b5b1 - depends: - - __osx >=11.0 - - libcxx >=18 - - numpy >=1.19,<3 - - numpy >=1.22.4 - - python >=3.12,<3.13.0a0 - - python >=3.12,<3.13.0a0 *_cpython - - python-dateutil >=2.8.2 - - python-tzdata >=2022.7 - - python_abi 3.12.* *_cp312 - - pytz >=2020.1 - constrains: - - xlrd >=2.0.1 - - pyxlsb >=1.0.10 - - pyreadstat >=1.2.0 - - fsspec >=2022.11.0 - - matplotlib >=3.6.3 - - s3fs >=2022.11.0 - - pyqt5 >=5.15.9 - - lxml >=4.9.2 - - blosc >=1.21.3 - - tabulate >=0.9.0 - - fastparquet >=2022.12.0 - - numba >=0.56.4 - - scipy >=1.10.0 - - xlsxwriter >=3.0.5 - - gcsfs >=2022.11.0 - - html5lib >=1.1 - - odfpy >=1.4.1 - - bottleneck >=1.3.6 - - numexpr >=2.8.4 - - beautifulsoup4 >=4.11.2 - - pyarrow >=10.0.1 - - openpyxl >=3.1.0 - - qtpy >=2.3.0 - - pytables >=3.8.0 - - tzdata >=2022.7 - - zstandard >=0.19.0 - - psycopg2 >=2.9.6 - - xarray >=2022.12.0 - - sqlalchemy >=2.0.0 - - python-calamine >=0.1.7 - - pandas-gbq >=0.19.0 - license: BSD-3-Clause - license_family: BSD - run_exports: {} - size: 14054660 - timestamp: 1749100309197 - conda: https://conda.anaconda.org/conda-forge/osx-arm64/pandas-2.3.1-py312h98f7732_0.conda sha256: f4f98436dde01309935102de2ded045bb5500b42fb30a3bf8751b15affee4242 md5: d3775e9b27579a0e96150ce28a2542bd @@ -13570,6 +13522,62 @@ packages: - pkg:pypi/pandas?source=hash-mapping size: 13991815 timestamp: 1752082557265 +- conda: https://conda.anaconda.org/conda-forge/osx-arm64/pandas-3.0.5-py312h6510ced_1.conda + sha256: 2cef45eb539621fb181b9682d960157a476729b61094046e44bb8ac167cf3779 + md5: 49c93ebce99da3235ba5bd1a24f7153a + depends: + - python + - numpy >=1.26.0 + - python-dateutil >=2.8.2 + - libcxx >=19 + - python 3.12.* *_cpython + - __osx >=11.0 + - python_abi 3.12.* *_cp312 + - numpy >=1.23,<3 + constrains: + - adbc-driver-postgresql >=1.2.0 + - adbc-driver-sqlite >=1.2.0 + - beautifulsoup4 >=4.12.3 + - blosc >=1.21.3 + - bottleneck >=1.4.2 + - fastparquet >=2024.11.0 + - fsspec >=2024.10.0 + - gcsfs >=2024.10.0 + - html5lib >=1.1 + - hypothesis >=6.116.0 + - jinja2 >=3.1.5 + - lxml >=5.3.0 + - matplotlib >=3.9.3 + - numba >=0.60.0 + - numexpr >=2.10.2 + - odfpy >=1.4.1 + - openpyxl >=3.1.5 + - psycopg2 >=2.9.10 + - pyarrow >=13.0.0 + - pyiceberg >=0.8.1 + - pymysql >=1.1.1 + - pyqt5 >=5.15.9 + - pyreadstat >=1.2.8 + - pytables >=3.10.1 + - pytest >=8.3.4 + - pytest-xdist >=3.6.1 + - python-calamine >=0.3.0 + - pytz >=2024.2 + - pyxlsb >=1.0.10 + - qtpy >=2.4.2 + - scipy >=1.14.1 + - s3fs >=2024.10.0 + - sqlalchemy >=2.0.36 + - tabulate >=0.9.0 + - xarray >=2024.10.0 + - xlrd >=2.0.1 + - xlsxwriter >=3.2.0 + - zstandard >=0.23.0 + license: BSD-3-Clause + license_family: BSD + run_exports: {} + size: 13962363 + timestamp: 1785276448997 - conda: https://conda.anaconda.org/conda-forge/osx-arm64/pango-1.56.4-h875632e_0.conda sha256: 705484ad60adee86cab1aad3d2d8def03a699ece438c864e8ac995f6f66401a6 md5: 7d57f8b4b7acfc75c777bc231f0d31be @@ -17886,57 +17894,6 @@ packages: - orc >=2.3.0,<2.3.1.0a0 size: 1438607 timestamp: 1773230254230 -- conda: https://conda.anaconda.org/conda-forge/win-64/pandas-2.3.0-py312h72972c8_0.conda - sha256: e4c8a685cfa1334a566b642523c9584d79ba78ed05888c7b7809d9116b6e9e25 - md5: e2ab2d8cc52281c9ebe19451936802eb - depends: - - numpy >=1.19,<3 - - numpy >=1.22.4 - - python >=3.12,<3.13.0a0 - - python-dateutil >=2.8.2 - - python-tzdata >=2022.7 - - python_abi 3.12.* *_cp312 - - pytz >=2020.1 - - ucrt >=10.0.20348.0 - - vc >=14.2,<15 - - vc14_runtime >=14.29.30139 - constrains: - - pyarrow >=10.0.1 - - gcsfs >=2022.11.0 - - fsspec >=2022.11.0 - - lxml >=4.9.2 - - tabulate >=0.9.0 - - openpyxl >=3.1.0 - - pyreadstat >=1.2.0 - - xlrd >=2.0.1 - - pyqt5 >=5.15.9 - - pyxlsb >=1.0.10 - - s3fs >=2022.11.0 - - zstandard >=0.19.0 - - numexpr >=2.8.4 - - python-calamine >=0.1.7 - - beautifulsoup4 >=4.11.2 - - fastparquet >=2022.12.0 - - bottleneck >=1.3.6 - - xarray >=2022.12.0 - - xlsxwriter >=3.0.5 - - sqlalchemy >=2.0.0 - - psycopg2 >=2.9.6 - - matplotlib >=3.6.3 - - blosc >=1.21.3 - - pytables >=3.8.0 - - html5lib >=1.1 - - numba >=0.56.4 - - tzdata >=2022.7 - - pandas-gbq >=0.19.0 - - qtpy >=2.3.0 - - scipy >=1.10.0 - - odfpy >=1.4.1 - license: BSD-3-Clause - license_family: BSD - run_exports: {} - size: 13859642 - timestamp: 1749100498003 - conda: https://conda.anaconda.org/conda-forge/win-64/pandas-2.3.1-py312hc128f0a_0.conda sha256: 711cf7b3aee4a92614744364ea996500b65fd5a11bceddb1fc03a5fd818b11d3 md5: 77e4ad6ddb37a0b489746352f8d2275d @@ -17989,6 +17946,63 @@ packages: - pkg:pypi/pandas?source=hash-mapping size: 13875687 timestamp: 1752082441874 +- conda: https://conda.anaconda.org/conda-forge/win-64/pandas-3.0.5-py312h95189c4_1.conda + sha256: c8895a5217e365ecad3c0572cfb3caace8cc2064b7de92ebe031ad1f2c315aab + md5: 2eebd79dac0d5cb03209bebf32c87e0a + depends: + - python + - numpy >=1.26.0 + - python-dateutil >=2.8.2 + - python-tzdata + - vc >=14.3,<15 + - vc14_runtime >=14.44.35208 + - ucrt >=10.0.20348.0 + - numpy >=1.23,<3 + - python_abi 3.12.* *_cp312 + constrains: + - adbc-driver-postgresql >=1.2.0 + - adbc-driver-sqlite >=1.2.0 + - beautifulsoup4 >=4.12.3 + - blosc >=1.21.3 + - bottleneck >=1.4.2 + - fastparquet >=2024.11.0 + - fsspec >=2024.10.0 + - gcsfs >=2024.10.0 + - html5lib >=1.1 + - hypothesis >=6.116.0 + - jinja2 >=3.1.5 + - lxml >=5.3.0 + - matplotlib >=3.9.3 + - numba >=0.60.0 + - numexpr >=2.10.2 + - odfpy >=1.4.1 + - openpyxl >=3.1.5 + - psycopg2 >=2.9.10 + - pyarrow >=13.0.0 + - pyiceberg >=0.8.1 + - pymysql >=1.1.1 + - pyqt5 >=5.15.9 + - pyreadstat >=1.2.8 + - pytables >=3.10.1 + - pytest >=8.3.4 + - pytest-xdist >=3.6.1 + - python-calamine >=0.3.0 + - pytz >=2024.2 + - pyxlsb >=1.0.10 + - qtpy >=2.4.2 + - scipy >=1.14.1 + - s3fs >=2024.10.0 + - sqlalchemy >=2.0.36 + - tabulate >=0.9.0 + - xarray >=2024.10.0 + - xlrd >=2.0.1 + - xlsxwriter >=3.2.0 + - zstandard >=0.23.0 + license: BSD-3-Clause + license_family: BSD + run_exports: {} + size: 13686325 + timestamp: 1785276278443 - conda: https://conda.anaconda.org/conda-forge/win-64/pango-1.56.4-h03d888a_0.conda sha256: dcda7e9bedc1c87f51ceef7632a5901e26081a1f74a89799a3e50dbdc801c0bd md5: 452d6d3b409edead3bd90fc6317cd6d4 diff --git a/workflow/envs/module.linux-64.pin.txt b/workflow/envs/module.linux-64.pin.txt index 95cf613..994bc64 100644 --- a/workflow/envs/module.linux-64.pin.txt +++ b/workflow/envs/module.linux-64.pin.txt @@ -47,13 +47,11 @@ https://conda.anaconda.org/conda-forge/linux-64/pydantic-core-2.46.4-py312h868fb https://conda.anaconda.org/conda-forge/noarch/annotated-types-0.7.0-pyhd8ed1ab_1.conda#2934f256a8acfe48f6ebb4fce6cde29c https://conda.anaconda.org/conda-forge/noarch/pydantic-2.13.4-pyhcf101f3_0.conda#729843edafc0899b3348bd3f19525b9d https://conda.anaconda.org/conda-forge/noarch/packaging-26.2-pyhc364b38_0.conda#4c06a92e74452cfa53623a81592e8934 -https://conda.anaconda.org/conda-forge/noarch/pandera-base-0.24.0-pyhd8ed1ab_2.conda#e1c50e117a98e39d297d9290132f032b -https://conda.anaconda.org/conda-forge/noarch/pytz-2026.2-pyhcf101f3_0.conda#03cb60f505ad3ada0a95277af5faeb1a -https://conda.anaconda.org/conda-forge/noarch/python-tzdata-2026.2-pyhd8ed1ab_0.conda#f6ad7450fc21e00ecc23812baed6d2e4 +https://conda.anaconda.org/conda-forge/noarch/pandera-base-0.32.1-pyhcf101f3_0.conda#c78daae943cb94e6e6b2eab16e16a18f https://conda.anaconda.org/conda-forge/noarch/six-1.17.0-pyhe01879c_1.conda#3339e3b65d58accf4ca4fb8748ab16b3 https://conda.anaconda.org/conda-forge/noarch/python-dateutil-2.9.0.post0-pyhe01879c_2.conda#5b8d21249ff20967101ffa321cab24e8 -https://conda.anaconda.org/conda-forge/linux-64/pandas-2.3.0-py312hf9745cd_0.conda#ac82ac336dbe61106e21fb2e11704459 -https://conda.anaconda.org/conda-forge/noarch/pandera-0.24.0-hd8ed1ab_2.conda#a12a21a89519f0cc224e44da86f5be2c +https://conda.anaconda.org/conda-forge/linux-64/pandas-3.0.5-py312h8ecdadd_1.conda#85eb29ade84d1bd97be5ec55607efb00 +https://conda.anaconda.org/conda-forge/noarch/pandera-0.32.1-ha00cc4c_0.conda#f6b96a51cda3cf9a068fd748c8155ced https://conda.anaconda.org/conda-forge/noarch/xyzservices-2026.3.0-pyhd8ed1ab_0.conda#4487b9c371d0161d54b5c7bbd890c0fc https://conda.anaconda.org/conda-forge/linux-64/sqlite-3.53.3-hbc0de68_0.conda#b345ea7f13e4cae9809c69b1d1af2c99 https://conda.anaconda.org/conda-forge/linux-64/libwebp-base-1.6.0-hd42ef1d_0.conda#aea31d2e5b1091feca96fcfe945c3cf9 @@ -131,7 +129,7 @@ https://conda.anaconda.org/conda-forge/noarch/joblib-1.5.3-pyhd8ed1ab_0.conda#61 https://conda.anaconda.org/conda-forge/linux-64/scikit-learn-1.9.0-np2py312h3226591_0.conda#e6e9b5795bb495325c3b4ebd451519aa https://conda.anaconda.org/conda-forge/noarch/networkx-3.6.1-pyhcf101f3_0.conda#a2c1eeadae7a309daed9d62c96012a2b https://conda.anaconda.org/conda-forge/noarch/mapclassify-2.10.0-pyhd8ed1ab_1.conda#cc293b4cad9909bf66ca117ea90d4631 -https://conda.anaconda.org/conda-forge/noarch/geopandas-base-1.0.1-pyha770c72_3.conda#e8343d1b635bf09dafdd362d7357f395 +https://conda.anaconda.org/conda-forge/noarch/geopandas-base-1.1.4-pyha770c72_0.conda#6e31007c02f361338281bb78c5b84e7c https://conda.anaconda.org/conda-forge/noarch/pysocks-1.7.1-pyha55dd90_7.conda#461219d1a5bd61342293efa2c0c90eac https://conda.anaconda.org/conda-forge/noarch/hyperframe-6.1.0-pyhd8ed1ab_0.conda#8e6923fc12f1fe8f8c4e5c9f343256ac https://conda.anaconda.org/conda-forge/noarch/hpack-4.2.0-pyhd8ed1ab_0.conda#b395909221b9bd1df066e5930e18855b @@ -146,8 +144,8 @@ https://conda.anaconda.org/conda-forge/linux-64/markupsafe-3.0.3-py312h8a5da7c_1 https://conda.anaconda.org/conda-forge/noarch/jinja2-3.1.6-pyhcf101f3_1.conda#04558c96691bed63104678757beb4f8d https://conda.anaconda.org/conda-forge/noarch/branca-0.8.2-pyhd8ed1ab_0.conda#1fcdf88e7a8c296d3df8409bf0690db4 https://conda.anaconda.org/conda-forge/noarch/folium-0.20.0-pyhd8ed1ab_0.conda#a6997a7dcd6673c0692c61dfeaea14ab -https://conda.anaconda.org/conda-forge/noarch/geopandas-1.0.1-pyhd8ed1ab_3.conda#1baca589eb35814a392eaad6d152447e -https://conda.anaconda.org/conda-forge/noarch/pandera-geopandas-0.24.0-hd8ed1ab_2.conda#fb21509f073465506ea994dc4667bf66 +https://conda.anaconda.org/conda-forge/noarch/geopandas-1.1.4-pyhd8ed1ab_0.conda#99d77e0f92f7b0b977f77cfb49c7eaf6 +https://conda.anaconda.org/conda-forge/noarch/pandera-geopandas-0.32.1-h9ad522b_0.conda#b6b27c3f219ce4d01c8ea31d435f263c https://conda.anaconda.org/conda-forge/noarch/et_xmlfile-2.0.0-pyhd8ed1ab_1.conda#71bf9646cbfabf3022c8da4b6b4da737 https://conda.anaconda.org/conda-forge/linux-64/openpyxl-3.1.5-py312h7f6eeab_3.conda#04ab345ef65b88bcbb8ac3d083427bfc https://conda.anaconda.org/conda-forge/linux-64/tornado-6.5.7-py312h4c3975b_0.conda#55f526c3fb5302a1ce922612348442e1 diff --git a/workflow/envs/module.osx-arm64.pin.txt b/workflow/envs/module.osx-arm64.pin.txt index 53bf53a..e4a4a6c 100644 --- a/workflow/envs/module.osx-arm64.pin.txt +++ b/workflow/envs/module.osx-arm64.pin.txt @@ -41,13 +41,11 @@ https://conda.anaconda.org/conda-forge/osx-arm64/pydantic-core-2.46.4-py312hb9d4 https://conda.anaconda.org/conda-forge/noarch/annotated-types-0.7.0-pyhd8ed1ab_1.conda#2934f256a8acfe48f6ebb4fce6cde29c https://conda.anaconda.org/conda-forge/noarch/pydantic-2.13.4-pyhcf101f3_0.conda#729843edafc0899b3348bd3f19525b9d https://conda.anaconda.org/conda-forge/noarch/packaging-26.2-pyhc364b38_0.conda#4c06a92e74452cfa53623a81592e8934 -https://conda.anaconda.org/conda-forge/noarch/pandera-base-0.24.0-pyhd8ed1ab_2.conda#e1c50e117a98e39d297d9290132f032b -https://conda.anaconda.org/conda-forge/noarch/pytz-2026.2-pyhcf101f3_0.conda#03cb60f505ad3ada0a95277af5faeb1a -https://conda.anaconda.org/conda-forge/noarch/python-tzdata-2026.2-pyhd8ed1ab_0.conda#f6ad7450fc21e00ecc23812baed6d2e4 +https://conda.anaconda.org/conda-forge/noarch/pandera-base-0.32.1-pyhcf101f3_0.conda#c78daae943cb94e6e6b2eab16e16a18f https://conda.anaconda.org/conda-forge/noarch/six-1.17.0-pyhe01879c_1.conda#3339e3b65d58accf4ca4fb8748ab16b3 https://conda.anaconda.org/conda-forge/noarch/python-dateutil-2.9.0.post0-pyhe01879c_2.conda#5b8d21249ff20967101ffa321cab24e8 -https://conda.anaconda.org/conda-forge/osx-arm64/pandas-2.3.0-py312hcb1e3ce_0.conda#d95b29a40430115d6aa817f70be5b5b1 -https://conda.anaconda.org/conda-forge/noarch/pandera-0.24.0-hd8ed1ab_2.conda#a12a21a89519f0cc224e44da86f5be2c +https://conda.anaconda.org/conda-forge/osx-arm64/pandas-3.0.5-py312h6510ced_1.conda#49c93ebce99da3235ba5bd1a24f7153a +https://conda.anaconda.org/conda-forge/noarch/pandera-0.32.1-ha00cc4c_0.conda#f6b96a51cda3cf9a068fd748c8155ced https://conda.anaconda.org/conda-forge/noarch/xyzservices-2026.3.0-pyhd8ed1ab_0.conda#4487b9c371d0161d54b5c7bbd890c0fc https://conda.anaconda.org/conda-forge/osx-arm64/sqlite-3.53.3-h85ec8f2_0.conda#b9df2fcb7e9eba945c1c946c438c4586 https://conda.anaconda.org/conda-forge/osx-arm64/zstd-1.5.7-hbf9d68e_6.conda#ab136e4c34e97f34fb621d2592a393d8 @@ -124,7 +122,7 @@ https://conda.anaconda.org/conda-forge/noarch/joblib-1.5.3-pyhd8ed1ab_0.conda#61 https://conda.anaconda.org/conda-forge/osx-arm64/scikit-learn-1.9.0-np2py312he5ca3e3_0.conda#b2b777cdad320a08da92bb1f58040bfd https://conda.anaconda.org/conda-forge/noarch/networkx-3.6.1-pyhcf101f3_0.conda#a2c1eeadae7a309daed9d62c96012a2b https://conda.anaconda.org/conda-forge/noarch/mapclassify-2.10.0-pyhd8ed1ab_1.conda#cc293b4cad9909bf66ca117ea90d4631 -https://conda.anaconda.org/conda-forge/noarch/geopandas-base-1.0.1-pyha770c72_3.conda#e8343d1b635bf09dafdd362d7357f395 +https://conda.anaconda.org/conda-forge/noarch/geopandas-base-1.1.4-pyha770c72_0.conda#6e31007c02f361338281bb78c5b84e7c https://conda.anaconda.org/conda-forge/noarch/pysocks-1.7.1-pyha55dd90_7.conda#461219d1a5bd61342293efa2c0c90eac https://conda.anaconda.org/conda-forge/noarch/hyperframe-6.1.0-pyhd8ed1ab_0.conda#8e6923fc12f1fe8f8c4e5c9f343256ac https://conda.anaconda.org/conda-forge/noarch/hpack-4.2.0-pyhd8ed1ab_0.conda#b395909221b9bd1df066e5930e18855b @@ -139,8 +137,8 @@ https://conda.anaconda.org/conda-forge/osx-arm64/markupsafe-3.0.3-py312h04c11ed_ https://conda.anaconda.org/conda-forge/noarch/jinja2-3.1.6-pyhcf101f3_1.conda#04558c96691bed63104678757beb4f8d https://conda.anaconda.org/conda-forge/noarch/branca-0.8.2-pyhd8ed1ab_0.conda#1fcdf88e7a8c296d3df8409bf0690db4 https://conda.anaconda.org/conda-forge/noarch/folium-0.20.0-pyhd8ed1ab_0.conda#a6997a7dcd6673c0692c61dfeaea14ab -https://conda.anaconda.org/conda-forge/noarch/geopandas-1.0.1-pyhd8ed1ab_3.conda#1baca589eb35814a392eaad6d152447e -https://conda.anaconda.org/conda-forge/noarch/pandera-geopandas-0.24.0-hd8ed1ab_2.conda#fb21509f073465506ea994dc4667bf66 +https://conda.anaconda.org/conda-forge/noarch/geopandas-1.1.4-pyhd8ed1ab_0.conda#99d77e0f92f7b0b977f77cfb49c7eaf6 +https://conda.anaconda.org/conda-forge/noarch/pandera-geopandas-0.32.1-h9ad522b_0.conda#b6b27c3f219ce4d01c8ea31d435f263c https://conda.anaconda.org/conda-forge/noarch/et_xmlfile-2.0.0-pyhd8ed1ab_1.conda#71bf9646cbfabf3022c8da4b6b4da737 https://conda.anaconda.org/conda-forge/osx-arm64/openpyxl-3.1.5-py312h2a925e6_3.conda#81e1c2e42e0bed7dc7d412dbb1b53a46 https://conda.anaconda.org/conda-forge/osx-arm64/tornado-6.5.7-py312h2bbb03f_0.conda#d037e9adb0365ab53445f357bd9a035f diff --git a/workflow/envs/module.win-64.pin.txt b/workflow/envs/module.win-64.pin.txt index b279923..439125a 100644 --- a/workflow/envs/module.win-64.pin.txt +++ b/workflow/envs/module.win-64.pin.txt @@ -46,13 +46,12 @@ https://conda.anaconda.org/conda-forge/win-64/pydantic-core-2.46.4-py312hdabe01f https://conda.anaconda.org/conda-forge/noarch/annotated-types-0.7.0-pyhd8ed1ab_1.conda#2934f256a8acfe48f6ebb4fce6cde29c https://conda.anaconda.org/conda-forge/noarch/pydantic-2.13.4-pyhcf101f3_0.conda#729843edafc0899b3348bd3f19525b9d https://conda.anaconda.org/conda-forge/noarch/packaging-26.2-pyhc364b38_0.conda#4c06a92e74452cfa53623a81592e8934 -https://conda.anaconda.org/conda-forge/noarch/pandera-base-0.24.0-pyhd8ed1ab_2.conda#e1c50e117a98e39d297d9290132f032b -https://conda.anaconda.org/conda-forge/noarch/pytz-2026.2-pyhcf101f3_0.conda#03cb60f505ad3ada0a95277af5faeb1a +https://conda.anaconda.org/conda-forge/noarch/pandera-base-0.32.1-pyhcf101f3_0.conda#c78daae943cb94e6e6b2eab16e16a18f https://conda.anaconda.org/conda-forge/noarch/python-tzdata-2026.2-pyhd8ed1ab_0.conda#f6ad7450fc21e00ecc23812baed6d2e4 https://conda.anaconda.org/conda-forge/noarch/six-1.17.0-pyhe01879c_1.conda#3339e3b65d58accf4ca4fb8748ab16b3 https://conda.anaconda.org/conda-forge/noarch/python-dateutil-2.9.0.post0-pyhe01879c_2.conda#5b8d21249ff20967101ffa321cab24e8 -https://conda.anaconda.org/conda-forge/win-64/pandas-2.3.0-py312h72972c8_0.conda#e2ab2d8cc52281c9ebe19451936802eb -https://conda.anaconda.org/conda-forge/noarch/pandera-0.24.0-hd8ed1ab_2.conda#a12a21a89519f0cc224e44da86f5be2c +https://conda.anaconda.org/conda-forge/win-64/pandas-3.0.5-py312h95189c4_1.conda#2eebd79dac0d5cb03209bebf32c87e0a +https://conda.anaconda.org/conda-forge/noarch/pandera-0.32.1-ha00cc4c_0.conda#f6b96a51cda3cf9a068fd748c8155ced https://conda.anaconda.org/conda-forge/noarch/xyzservices-2026.3.0-pyhd8ed1ab_0.conda#4487b9c371d0161d54b5c7bbd890c0fc https://conda.anaconda.org/conda-forge/win-64/sqlite-3.53.3-hdb435a2_0.conda#7a28cb97617f0bd6a650b4b88f1f44cb https://conda.anaconda.org/conda-forge/win-64/zstd-1.5.7-h534d264_6.conda#053b84beec00b71ea8ff7a4f84b55207 @@ -122,7 +121,7 @@ https://conda.anaconda.org/conda-forge/noarch/joblib-1.5.3-pyhd8ed1ab_0.conda#61 https://conda.anaconda.org/conda-forge/win-64/scikit-learn-1.9.0-np2py312hea30aaf_0.conda#0637562978d4231902754645e4c28640 https://conda.anaconda.org/conda-forge/noarch/networkx-3.6.1-pyhcf101f3_0.conda#a2c1eeadae7a309daed9d62c96012a2b https://conda.anaconda.org/conda-forge/noarch/mapclassify-2.10.0-pyhd8ed1ab_1.conda#cc293b4cad9909bf66ca117ea90d4631 -https://conda.anaconda.org/conda-forge/noarch/geopandas-base-1.0.1-pyha770c72_3.conda#e8343d1b635bf09dafdd362d7357f395 +https://conda.anaconda.org/conda-forge/noarch/geopandas-base-1.1.4-pyha770c72_0.conda#6e31007c02f361338281bb78c5b84e7c https://conda.anaconda.org/conda-forge/noarch/win_inet_pton-1.1.0-pyh7428d3b_8.conda#46e441ba871f524e2b067929da3051c2 https://conda.anaconda.org/conda-forge/noarch/pysocks-1.7.1-pyh09c184e_7.conda#e2fd202833c4a981ce8a65974fe4abd1 https://conda.anaconda.org/conda-forge/noarch/hyperframe-6.1.0-pyhd8ed1ab_0.conda#8e6923fc12f1fe8f8c4e5c9f343256ac @@ -138,8 +137,8 @@ https://conda.anaconda.org/conda-forge/win-64/markupsafe-3.0.3-py312h05f76fc_1.c https://conda.anaconda.org/conda-forge/noarch/jinja2-3.1.6-pyhcf101f3_1.conda#04558c96691bed63104678757beb4f8d https://conda.anaconda.org/conda-forge/noarch/branca-0.8.2-pyhd8ed1ab_0.conda#1fcdf88e7a8c296d3df8409bf0690db4 https://conda.anaconda.org/conda-forge/noarch/folium-0.20.0-pyhd8ed1ab_0.conda#a6997a7dcd6673c0692c61dfeaea14ab -https://conda.anaconda.org/conda-forge/noarch/geopandas-1.0.1-pyhd8ed1ab_3.conda#1baca589eb35814a392eaad6d152447e -https://conda.anaconda.org/conda-forge/noarch/pandera-geopandas-0.24.0-hd8ed1ab_2.conda#fb21509f073465506ea994dc4667bf66 +https://conda.anaconda.org/conda-forge/noarch/geopandas-1.1.4-pyhd8ed1ab_0.conda#99d77e0f92f7b0b977f77cfb49c7eaf6 +https://conda.anaconda.org/conda-forge/noarch/pandera-geopandas-0.32.1-h9ad522b_0.conda#b6b27c3f219ce4d01c8ea31d435f263c https://conda.anaconda.org/conda-forge/noarch/et_xmlfile-2.0.0-pyhd8ed1ab_1.conda#71bf9646cbfabf3022c8da4b6b4da737 https://conda.anaconda.org/conda-forge/win-64/openpyxl-3.1.5-py312h83acffa_3.conda#ff342a314798173eaaf2753a22f044fa https://conda.anaconda.org/conda-forge/win-64/tornado-6.5.7-py312he06e257_0.conda#1045d29f787812d3fac1fd80a1339710 diff --git a/workflow/internal/config.schema.yaml b/workflow/internal/config.schema.yaml index bbc98c0..4ad8df4 100644 --- a/workflow/internal/config.schema.yaml +++ b/workflow/internal/config.schema.yaml @@ -59,8 +59,8 @@ $defs: title: Planned commissioning year windows description: | Technology- and status-specific year-offset windows used to impute - missing commissioning years for planned powerplants. Each window is - expressed as [minimum_year_offset, maximum_year_offset] relative to + missing commissioning years for planned powerplants. + Each window is expressed as [minimum_year_offset, maximum_year_offset) relative to the dataset year (inclusive). type: object additionalProperties: @@ -460,7 +460,7 @@ properties: additionalProperties: false required: - scenario - - start_year_imputation_method + - method - lifetime_years - retirement_delay_years - planned_commissioning_year_windows @@ -483,14 +483,14 @@ properties: $ref: "#/$defs/year_map" retirement_delay_years: $ref: "#/$defs/year_map" - start_year_imputation_method: - title: Start year imputation method + method: + title: Time imputation method description: | - Method used to impute missing start years when backfilling from known - end years is not possible. + Method used to complete powerplants for which neither a start nor end + year is known. - - 'capacity_profile' assigns missing start years deterministically using - capacity-weighted commissioning profiles from annual commissioning statistics. + - 'capacity_profile' assigns dates deterministically using capacity-weighted + commissioning and retirement profiles from annual capacity statistics. type: string enum: - capacity_profile diff --git a/workflow/rules/impute.smk b/workflow/rules/impute.smk index 155d8b8..f6b25c9 100644 --- a/workflow/rules/impute.smk +++ b/workflow/rules/impute.smk @@ -88,7 +88,7 @@ rule impute_time: message: "National-level imputation of missing powerplant ages in {wildcards.shapes}-{wildcards.category} dataset." script: - "../scripts/impute_ages.py" + "../scripts/impute_time.py" rule impute_capacity_adjustment: diff --git a/workflow/scripts/_plots.py b/workflow/scripts/_plots.py index 36ed92d..98e6610 100644 --- a/workflow/scripts/_plots.py +++ b/workflow/scripts/_plots.py @@ -12,6 +12,7 @@ from matplotlib import ticker as mticker from matplotlib.axes import Axes from matplotlib.patches import Patch +from matplotlib.typing import ColorType def draw_empty(ax: Axes, title: str = "", message="No data available"): @@ -248,7 +249,7 @@ def get_colour_dict( colormap: str, *, value_range: tuple[float, float] = (0.0, 1.0), -) -> dict[str, str]: +) -> dict[str, ColorType]: """Return deterministic colours for a collection of source types.""" sorted_sources = sorted(sources) diff --git a/workflow/scripts/_schemas.py b/workflow/scripts/_schemas.py index 2951033..a369d5f 100644 --- a/workflow/scripts/_schemas.py +++ b/workflow/scripts/_schemas.py @@ -1,17 +1,31 @@ """Schemas for key files.""" -# ruff: noqa: UP007 from typing import Literal import _utils +import pandas as pd +import pandera.geopandas as gpa import pandera.pandas as pa -from pandera.pandas import DataFrameModel, Field, check from pandera.typing.geopandas import GeoSeries from pandera.typing.pandas import Index, Series from shapely.geometry import Point +# A diverse set of statuses to diminish oversimplification during gap filling. +OPERATING = "operating" +RETIRED = "retired" +HISTORICAL = {OPERATING, RETIRED} +PLANNED = {"construction", "pre-construction", "announced"} +SCENARIO_MAP = { + "historical": HISTORICAL, + "construction": HISTORICAL | {"construction"}, + "pre_construction": HISTORICAL | {"construction", "pre-construction"}, + "announced": HISTORICAL | PLANNED, +} +# Status categorisation shown to users (and accepted in user imputed files). +IMPUTED_STATUS = {"planned", "operating", "retired"} + -class EIASchema(DataFrameModel): +class EIASchema(pa.DataFrameModel): class Config: coerce = True strict = True @@ -20,57 +34,57 @@ class Config: "Sample year" category: Series[str] "Human readable name" - capacity_mw: Series[float] = Field(nullable=True) + capacity_mw: Series[float] = pa.Field(ge=0, nullable=True) "Electrical capacity in Megawatt" country_id: Series[str] "Country ISO-3 code" -class ShapeSchema(DataFrameModel): +class ShapeSchema(gpa.GeoDataFrameModel): class Config: coerce = True strict = False - shape_id: Series[str] = Field(unique=True) + shape_id: Series[str] = gpa.Field(unique=True) "Unique ID for this shape." country_id: Series[str] "ISO alpha-3 code." - shape_class: Series[str] = Field(isin=["land", "maritime"]) + shape_class: Series[str] = gpa.Field(isin=["land", "maritime"]) "Shape classifier" geometry: GeoSeries "Shape polygon." - @check("geometry", element_wise=True) + @gpa.check("geometry", element_wise=True) def geom_not_empty(cls, geom): return (geom is not None) and (not geom.is_empty) and geom.is_valid -class AggregatedPlantSchema(DataFrameModel): +class AggregatedPlantSchema(pa.DataFrameModel): class Config: coerce = True strict = True - index: Index[int] = Field(unique=True) + index: Index[int] = pa.Field(unique=True) shape_id: Series[str] country_id: Series[str] category: Series[str] technology: Series[str] - output_capacity_mw: Series[float] = Field(gt=0) + output_capacity_mw: Series[float] = pa.Field(gt=0) chp: Series[bool] | None ccs: Series[bool] | None fuel_class: Series[str] | None -class PlantSchema(DataFrameModel): +class PlantSchema(gpa.GeoDataFrameModel): class Config: coerce = True strict = True - index: Index[int] = Field(unique=True) + index: Index[int] = gpa.Field(unique=True) # Identifiers - powerplant_id: Series[str] = Field(unique=True) + powerplant_id: Series[str] = gpa.Field(unique=True) "Unique ID for the powerplant." name: Series[str] "Human readable powerplant name." @@ -79,7 +93,7 @@ class Config: "General category of the powerplant." technology: Series[str] "Subcategory of the powerplant, if necessary." - output_capacity_mw: Series[float] = Field(gt=0) + output_capacity_mw: Series[float] = gpa.Field(gt=0) "Powerplant gross output capacity in Megawatts." # Temporal aspects start_year: Series[float] @@ -88,18 +102,20 @@ class Config: "Expected decommissioning year." status: Series[str] "Known state of the project." - start_year_source_type: Series[str] | None = Field( + start_year_source_type: Series[str] | None = gpa.Field( isin=_utils.date_source_types_for("start_year") ) "Source/provenance label for the start year." - end_year_source_type: Series[str] | None = Field( + end_year_source_type: Series[str] | None = gpa.Field( isin=_utils.date_source_types_for("end_year") ) "Source/provenance label for the end year." # Location / size - geometry: GeoSeries[Point] = Field() + geometry: GeoSeries[Point] = gpa.Field() "Powerplant point data." - country_id: Series[str] | None = Field(str_length={"min_value": 3, "max_value": 3}) + country_id: Series[str] | None = gpa.Field( + str_length={"min_value": 3, "max_value": 3} + ) # Combustion specifics ccs: Series[bool] | None """Identifier for known CCS-enabled powerplants.""" @@ -108,15 +124,21 @@ class Config: fuel_class: Series[str] | None """Unique ID in the fuel consumption look-up table.""" # Hydropower specifics - reservoir_km3: Series[float] | None = Field(nullable=True, ge=0) + reservoir_km3: Series[float] | None = gpa.Field(nullable=True, ge=0) """Reservoir volume.""" - @check("geometry", element_wise=True) + @gpa.check("geometry", element_wise=True) def geom_not_empty(cls, geom): return (geom is not None) and (not geom.is_empty) and geom.is_valid + @gpa.dataframe_check + def end_after_start(cls, plants: pd.DataFrame): + """Require ordered dates wherever both years are known.""" + known_dates = plants[["start_year", "end_year"]].notna().all(axis="columns") + return ~known_dates | plants["end_year"].gt(plants["start_year"]) + -class FuelSchema(DataFrameModel): +class FuelSchema(pa.DataFrameModel): class Config: strict = True coerce = True @@ -127,16 +149,25 @@ class Config: "Fuel consumed." -# A diverse set of statuses to diminish oversimplification during gap filling. -PREPARED_STATUS = { - "announced", - "pre-construction", - "construction", - "operating", - "retired", -} -# Status categorisation shown to users (and accepted in user imputed files). -IMPUTED_STATUS = {"planned", "operating", "retired"} +class CapacityDateEventSchema(pa.DataFrameModel): + """Annual commissioning and retirement events used by diagnostics.""" + + class Config: + coerce = True + strict = True + + powerplant_id: Series[str] + name: Series[str] + country_id: Series[str] = pa.Field(str_length={"min_value": 3, "max_value": 3}) + category: Series[str] + technology: Series[str] + status: Series[str] = pa.Field(isin={"planned", "operating", "retired"}) + year: Series[float] + event_type: Series[str] = pa.Field(isin={"commissioning", "retirement"}) + source_type: Series[str] = pa.Field(isin=set(_utils.DATE_SOURCE_METADATA)) + source_label: Series[str] + output_capacity_mw: Series[float] = pa.Field(gt=0) + capacity_change_mw: Series[float] = pa.Field(ne=0) def build_schema( @@ -145,7 +176,7 @@ def build_schema( """Construct an inflexible schema applicable to each processing stage.""" schema = PlantSchema.to_schema() if stage == "prepare": - status_set = PREPARED_STATUS + status_set = SCENARIO_MAP["announced"] # Years can be empty during preparation stages year_overrides: list[tuple[str, dict]] = [ ("start_year", {"nullable": True}), diff --git a/workflow/scripts/_utils.py b/workflow/scripts/_utils.py index 7743b64..e284eff 100644 --- a/workflow/scripts/_utils.py +++ b/workflow/scripts/_utils.py @@ -1,6 +1,6 @@ """General utilities shared across rules.""" -from typing import Literal +from typing import Literal, TypedDict import geopandas as gpd import pandas as pd @@ -38,18 +38,25 @@ def listify(item) -> list: return item if is_list_like(item) else [item] -EIA_CAT_MAPPING = { - "bioenergy": "biomass and waste", - "fossil": "fossil fuels", - "geothermal": "geothermal", +EIA_CAT_MAPPING: dict[str, list[str]] = { + "bioenergy": ["biomass and waste"], + "fossil": ["fossil fuels"], + "geothermal": ["geothermal"], "hydropower": ["hydropower", "pumped storage"], - "nuclear": "nuclear", - "solar": "solar", - "wind": "wind", + "nuclear": ["nuclear"], + "solar": ["solar"], + "wind": ["wind"], } -EIA_CAT_MAPPING = {k: listify(v) for k, v in EIA_CAT_MAPPING.items()} -DATE_SOURCE_METADATA = { + +class DateSourceMetadata(TypedDict): + """Display metadata for an imputed-date source type.""" + + label: str + applies_to: set[str] + + +DATE_SOURCE_METADATA: dict[str, DateSourceMetadata] = { "observed": { "label": "Observed date (from powerplant data)", "applies_to": {"start_year", "end_year"}, @@ -74,8 +81,8 @@ def listify(item) -> list: "label": "Start date imputed within announced window", "applies_to": {"start_year"}, }, - "derived_from_imputed_retirement_end_year": { - "label": "Start date derived from retirement-profile end date", + "imputed_capacity_profile_retirement_linked": { + "label": "Commissioning profile (retirement-linked)", "applies_to": {"start_year"}, }, "derived_from_start_year_lifetime": { @@ -94,10 +101,6 @@ def listify(item) -> list: "label": "End date derived from start date, lifetime, and retirement delay", "applies_to": {"end_year"}, }, - "observed_adjusted_with_retirement_delay": { - "label": "Observed end date adjusted with retirement delay", - "applies_to": {"end_year"}, - }, } @@ -147,7 +150,7 @@ def get_combined_text_col( Form: {prefix}col1{sep}col2{sep}...coln{suffix}. """ - return prefix + raw[cols].astype(str).agg(sep.join, axis="columns") + suffix + return prefix + raw[cols].fillna("").map(str).agg(sep.join, axis="columns") + suffix def check_single_category(df: pd.DataFrame) -> str: @@ -201,6 +204,14 @@ def ensure_positive_capacity(df: pd.DataFrame) -> pd.DataFrame: return df[df["output_capacity_mw"] > 0].copy() +def filter_noncontributing_powerplants(df: pd.DataFrame) -> pd.DataFrame: + """Remove source records with no positive capacity or operating duration.""" + filtered = ensure_positive_capacity(df) + known_dates = filtered[["start_year", "end_year"]].notna().all(axis="columns") + zero_duration = known_dates & filtered["start_year"].eq(filtered["end_year"]) + return filtered.loc[~zero_duration].copy() + + def get_adjusted_capacity( operating_plants: pd.DataFrame, expected_capacity: pd.Series ) -> pd.Series: diff --git a/workflow/scripts/impute_ages.py b/workflow/scripts/impute_ages.py deleted file mode 100644 index 57c878e..0000000 --- a/workflow/scripts/impute_ages.py +++ /dev/null @@ -1,1579 +0,0 @@ -"""Imputation of missing values.""" - -import math -import sys -from typing import TYPE_CHECKING, Any - -import _plots -import _schemas -import _utils -import geopandas as gpd -import numpy as np -import pandas as pd -from cmap import Colormap -from matplotlib import pyplot as plt -from matplotlib.lines import Line2D -from matplotlib.patches import Patch - -if TYPE_CHECKING: - snakemake: Any - -# Status Setup -OPERATING = "operating" -RETIRED = "retired" -HISTORICAL = {OPERATING, RETIRED} -PLANNED = {"construction", "pre-construction", "announced"} - -SCENARIO_MAP = { - "historical": HISTORICAL, - "construction": HISTORICAL | {"construction"}, - "pre_construction": HISTORICAL | {"construction", "pre-construction"}, - "announced": HISTORICAL | PLANNED, -} - -# Harmonise powerplant categories with the category names used by the -# annual reference-capacity dataset. -REFERENCE_CATEGORY_MAP = _utils.EIA_CAT_MAPPING - - -def _reference_categories(category: str) -> list[str]: - """Return reference-capacity categories matching a powerplant category.""" - return _utils.listify(REFERENCE_CATEGORY_MAP.get(category, category)) - - -def _reference_capacity_stock( - reference_capacity_df: pd.DataFrame, - country_id: str, - categories: list[str], - max_year: int, -) -> pd.Series: - """Return annual reference-capacity stock for mapped categories. - - Multiple reference categories may map to one powerplant category, such as - hydropower and pumped storage. These are summed by year before deriving - commissioning or retirement profiles. - """ - return ( - reference_capacity_df.loc[ - reference_capacity_df["country_id"].eq(country_id) - & reference_capacity_df["category"].isin(categories) - & reference_capacity_df["year"].le(max_year), - ["year", "capacity_mw"], - ] - .groupby("year", as_index=True)["capacity_mw"] - .sum(min_count=1) - .sort_index() - ) - - -def _reference_first_year( - reference_capacity_df: pd.DataFrame, - country_id: str, - categories: list[str], - max_year: int, -) -> int: - """Return first year with reported reference-capacity stock.""" - capacity_stock = _reference_capacity_stock( - reference_capacity_df=reference_capacity_df, - country_id=country_id, - categories=categories, - max_year=max_year, - ) - - reported_years = capacity_stock.loc[capacity_stock.notna()].index - - if reported_years.empty: - raise ValueError( - "No reported reference capacity years found for " - f"{country_id=} and {categories=} up to {max_year=}." - ) - - return int(reported_years.min()) - - -def _initial_year_source_type(year: pd.Series) -> pd.Series: - """Label whether year values were originally present or missing.""" - source_type = pd.Series("observed", index=year.index, dtype="object") - source_type.loc[year.isna()] = "missing_unresolved" - return source_type - - -def _build_reference_addition_profile( - reference_capacity_df: pd.DataFrame, - country_id: str, - categories: list[str], - years: pd.Index, - smoothing_window: int = 1, -) -> pd.DataFrame: - """Build an annual commissioning profile from capacity stock data.""" - capacity_stock = _reference_capacity_stock( - reference_capacity_df=reference_capacity_df, - country_id=country_id, - categories=categories, - max_year=_utils.DATASET_YEAR, - ) - - # Positive annual stock changes provide the temporal commissioning - # profile. Negative changes represent retirements or revisions and do - # not contribute commissioning weight hence clipped to 0. - reference_stock = capacity_stock.reindex(years) - reference_positive_change = ( - capacity_stock.diff().clip(lower=0.0).reindex(years).fillna(0.0) - ) - - profile_basis = reference_positive_change.copy() - - if smoothing_window > 1: - profile_basis = profile_basis.rolling( - window=smoothing_window, center=True, min_periods=1 - ).mean() - - profile_fallback_used = profile_basis.sum() <= 0 - - # If the reference series contains no positive capacity changes, there is no - # commissioning pattern to follow. Use a uniform profile so missing capacity can - # still be allocated across the candidate years. - if profile_fallback_used: - profile_basis.loc[:] = 1.0 - - profile = pd.DataFrame( - { - "reference_stock_mw": reference_stock, - "reference_positive_change_mw": reference_positive_change, - "reference_profile_basis_mw": profile_basis, - "reference_profile_weight": profile_basis / profile_basis.sum(), - }, - index=years, - ) - profile["profile_fallback_used"] = profile_fallback_used - - return profile - - -def _build_reference_retirement_profile( - reference_capacity_df: pd.DataFrame, - country_id: str, - categories: list[str], - years: pd.Index, -) -> pd.DataFrame: - """Build an annual retirement profile from capacity-stock reductions.""" - capacity_stock = _reference_capacity_stock( - reference_capacity_df=reference_capacity_df, - country_id=country_id, - categories=categories, - max_year=_utils.DATASET_YEAR - 1, - ) - - reference_stock = capacity_stock.reindex(years) - - # Negative annual stock changes provide the retirement profile. - # Positive changes represent net additions and carry no retirement weight. - reference_negative_change = ( - (-capacity_stock.diff()).clip(lower=0.0).reindex(years).fillna(0.0) - ) - - profile_basis = reference_negative_change.copy() - profile_fallback_used = profile_basis.sum() <= 0 - - # If the reference series contains no negative capacity changes, there is no - # retirement pattern to follow. Use a uniform profile so missing retirements can - # still be allocated across the candidate years. - if profile_fallback_used: - profile_basis.loc[:] = 1.0 - - profile = pd.DataFrame( - { - "reference_stock_mw": reference_stock, - "reference_negative_change_mw": reference_negative_change, - "reference_profile_basis_mw": profile_basis, - "reference_profile_weight": (profile_basis / profile_basis.sum()), - }, - index=years, - ) - profile["profile_fallback_used"] = profile_fallback_used - - return profile - - -def _build_clipped_residual_profile( - dated_df: pd.DataFrame, - profile: pd.DataFrame, - allocatable_years: pd.Index, - missing_capacity_mw: float, - year_col: str, -) -> pd.DataFrame: - """Add observed capacity and a feasible residual target to a profile.""" - result = profile.copy() - - observed_capacity = ( - dated_df.groupby(year_col)["output_capacity_mw"] - .sum() - .reindex(result.index, fill_value=0.0) - ) - - # The reference data determine the temporal shape, whilst the plant - # dataset determines the total capacity represented by the profile. - total_capacity_mw = observed_capacity.sum() + missing_capacity_mw - - result["observed_mw"] = observed_capacity - result["target_final_mw"] = result["reference_profile_weight"] * total_capacity_mw - result["raw_residual_mw"] = result["target_final_mw"] - result["observed_mw"] - - # Observed dates remain fixed. Years that already exceed the scaled - # target therefore receive no additional imputed capacity. - residual_target = ( - result["raw_residual_mw"] - .clip(lower=0.0) - .reindex(allocatable_years, fill_value=0.0) - ) - - residual_fallback_used = residual_target.sum() <= 0 - - if residual_fallback_used: - residual_target = result["reference_profile_weight"].reindex( - allocatable_years, fill_value=0.0 - ) - - if residual_target.sum() <= 0: - residual_target = pd.Series(1.0, index=allocatable_years, dtype=float) - - # Rescale the feasible residual so that all missing plant capacity is - # allocated while preserving its relative annual shape. - residual_target = residual_target / residual_target.sum() * missing_capacity_mw - - result["residual_target_mw"] = residual_target.reindex(result.index, fill_value=0.0) - result["residual_fallback_used"] = residual_fallback_used - - return result - - -def _select_largest_deficit_year( - remaining_target: pd.Series, - original_target: pd.Series, - *, - prefer_later_years: bool = False, -) -> int: - """Select the year with the largest remaining deficit using deterministic ties.""" - if remaining_target.empty: - raise ValueError( - "Cannot select an imputation year from an empty target profile." - ) - - target_ranking = pd.DataFrame( - { - "year": remaining_target.index, - "remaining_target": remaining_target.to_numpy(), - "original_target": original_target.loc[remaining_target.index].to_numpy(), - } - ).sort_values( - ["remaining_target", "original_target", "year"], - ascending=[False, False, not prefer_later_years], - ) - - return int(target_ranking.iloc[0]["year"]) - - -def _allocate_start_years_by_residual_target( - undated_df: pd.DataFrame, residual_target: pd.Series, lifetimes: dict[str, int] -) -> pd.Series: - """Assign whole plants to feasible years with the largest deficits. - - Plants with the narrowest feasible commissioning windows are processed - first. Within equal feasible windows, larger plants are processed first - because they are harder to fit into the residual target. - """ - capacities = undated_df["output_capacity_mw"] - earliest_feasible_year = _utils.DATASET_YEAR - undated_df["technology"].map( - lifetimes - ) - - # Allocate the least flexible plants first, then the largest capacity - # blocks, to reduce poor fits caused by indivisible plants. - order = pd.DataFrame( - { - "earliest_feasible_year": earliest_feasible_year, - "capacity": capacities, - "powerplant_id": undated_df["powerplant_id"], - "row_order": np.arange(len(undated_df)), - }, - index=undated_df.index, - ).sort_values( - ["earliest_feasible_year", "capacity", "powerplant_id", "row_order"], - ascending=[False, False, True, True], - ) - - remaining_target = residual_target.copy() - assigned_years = pd.Series(np.nan, index=undated_df.index, dtype=float) - - for plant_index in order.index: - plant = undated_df.loc[plant_index] - plant_capacity = capacities.loc[plant_index] - lifetime_years = lifetimes[plant["technology"]] - - earliest_year = _utils.DATASET_YEAR - lifetime_years - feasible_target = remaining_target.loc[ - (remaining_target.index >= earliest_year) - & (remaining_target.index <= _utils.DATASET_YEAR) - ] - - # Assign the plant to the feasible year with the largest remaining - # deficit. Original target size and year provide deterministic ties. - assigned_year = _select_largest_deficit_year( - remaining_target=feasible_target, original_target=residual_target - ) - - assigned_years.loc[plant_index] = assigned_year - remaining_target.loc[assigned_year] -= plant_capacity - - return assigned_years - - -def _allocate_years_by_target( - undated_df: pd.DataFrame, target: pd.Series, *, prefer_later_years: bool = False -) -> pd.Series: - """Assign whole plants to years with the largest remaining deficits.""" - order = ( - undated_df[["output_capacity_mw", "powerplant_id"]] - .assign(row_order=np.arange(len(undated_df))) - .sort_values( - ["output_capacity_mw", "powerplant_id", "row_order"], - ascending=[False, True, True], - ) - ) - - remaining_target = target.copy() - assigned_years = pd.Series(np.nan, index=undated_df.index, dtype=float) - - for plant_index in order.index: - plant_capacity = undated_df.loc[plant_index, "output_capacity_mw"] - - assigned_year = _select_largest_deficit_year( - remaining_target=remaining_target, - original_target=target, - prefer_later_years=prefer_later_years, - ) - - assigned_years.loc[plant_index] = assigned_year - remaining_target.loc[assigned_year] -= plant_capacity - - return assigned_years - - -def _capacity_by_assigned_year( - assigned_years: pd.Series, capacities: pd.Series, years: pd.Index, year_col: str -) -> pd.Series: - """Return assigned plant capacity aggregated to the requested years.""" - assignments = pd.DataFrame( - {year_col: assigned_years, "output_capacity_mw": capacities}, - index=assigned_years.index, - ) - - return ( - assignments.groupby(year_col)["output_capacity_mw"] - .sum() - .reindex(years, fill_value=0.0) - ) - - -def _complete_capacity_profile( - profile: pd.DataFrame, - undated_df: pd.DataFrame, - assigned_years: pd.Series, - country_id: str, - category: str, - reference_category: str, -) -> pd.DataFrame: - """Return the commissioning-profile target used in the diagnostic plot.""" - imputed_capacity = _capacity_by_assigned_year( - assigned_years=assigned_years, - capacities=undated_df["output_capacity_mw"], - years=profile.index, - year_col="start_year", - ) - - return pd.DataFrame( - { - "country_id": country_id, - "category": category, - "reference_category": reference_category, - "year": profile.index, - "target_final_mw": profile["target_final_mw"], - "observed_mw": profile["observed_mw"], - "imputed_mw": imputed_capacity, - } - ) - - -def _complete_retirement_profile( - profile: pd.DataFrame, - undated_df: pd.DataFrame, - assigned_end_years: pd.Series, - country_id: str, - category: str, - reference_category: str, -) -> pd.DataFrame: - """Return the retirement-profile target used in the diagnostic plot.""" - imputed_capacity = _capacity_by_assigned_year( - assigned_years=assigned_end_years, - capacities=undated_df["output_capacity_mw"], - years=profile.index, - year_col="end_year", - ) - - return pd.DataFrame( - { - "country_id": country_id, - "category": category, - "reference_category": reference_category, - "profile_source": profile["profile_source"], - "year": profile.index, - "target_final_mw": profile["target_final_mw"], - "observed_mw": profile["observed_mw"], - "imputed_mw": imputed_capacity, - } - ) - - -def _complete_planned_commissioning_profile( - undated_df: pd.DataFrame, - assigned_years: pd.Series, - target: pd.Series, - country_id: str, - category: str, - technology: str, - status: str, -) -> pd.DataFrame: - """Return the planned commissioning profile used in the diagnostic plot.""" - imputed_capacity = _capacity_by_assigned_year( - assigned_years=assigned_years, - capacities=undated_df["output_capacity_mw"], - years=target.index, - year_col="year", - ) - - return pd.DataFrame( - { - "country_id": country_id, - "category": category, - "technology": technology, - "status": status, - "year": target.index, - "target_imputed_mw": target, - "imputed_mw": imputed_capacity, - } - ) - - -def _impute_start_years_by_capacity_profile( - prepared_df: pd.DataFrame, - reference_capacity_df: pd.DataFrame, - lifetimes: dict[str, int], - smoothing_window: int = 1, -) -> tuple[pd.Series, pd.DataFrame]: - """Impute operating plant dates against a reference capacity profile.""" - start_year = prepared_df["start_year"].copy() - lifetime = prepared_df["technology"].map(lifetimes) - - result = pd.Series(np.nan, index=prepared_df.index, dtype=float) - profiles = [] - - missing_mask = start_year.isna() & prepared_df["status"].eq(OPERATING) - - if not missing_mask.any(): - return result, pd.DataFrame() - - grouped_missing = prepared_df.loc[missing_mask].groupby( - ["country_id", "category"], dropna=False - ) - - for (country_id, category), undated_group in grouped_missing: - reference_categories = _reference_categories(category) - reference_category_label = "+".join(reference_categories) - - group_mask = ( - prepared_df["country_id"].eq(country_id) - & prepared_df["category"].eq(category) - & prepared_df["status"].isin(HISTORICAL) - ) - - dated_group = prepared_df.loc[group_mask & start_year.notna()].copy() - dated_group["start_year"] = start_year.loc[dated_group.index] - - # this int() is necessary as the subtraction creates a float, was causing errors - earliest_allocatable_year = int( - (_utils.DATASET_YEAR - lifetime.loc[undated_group.index]).min() - ) - - allocatable_years = pd.Index( - range(earliest_allocatable_year, _utils.DATASET_YEAR + 1), name="start_year" - ) - - # this int() is necessary, was causing errors - reference_first_year = _reference_first_year( - reference_capacity_df=reference_capacity_df, - country_id=country_id, - categories=reference_categories, - max_year=_utils.DATASET_YEAR, - ) - - # the int() is necessary, was causing errors - observed_first_year = ( - int(dated_group["start_year"].min()) - if not dated_group.empty - else _utils.DATASET_YEAR - ) - - first_profile_year = min( - reference_first_year, observed_first_year, earliest_allocatable_year - ) - - profile_years = pd.Index( - range(first_profile_year, _utils.DATASET_YEAR + 1), name="start_year" - ) - - reference_profile = _build_reference_addition_profile( - reference_capacity_df=reference_capacity_df, - country_id=country_id, - categories=reference_categories, - years=profile_years, - smoothing_window=smoothing_window, - ) - - missing_capacity_mw = undated_group["output_capacity_mw"].sum() - - allocation_profile = _build_clipped_residual_profile( - dated_df=dated_group, - profile=reference_profile, - allocatable_years=allocatable_years, - missing_capacity_mw=missing_capacity_mw, - year_col="start_year", - ) - - assigned_years = _allocate_start_years_by_residual_target( - undated_df=undated_group, - residual_target=allocation_profile["residual_target_mw"], - lifetimes=lifetimes, - ) - - result.loc[undated_group.index] = assigned_years - - profiles.append( - _complete_capacity_profile( - profile=allocation_profile, - undated_df=undated_group, - assigned_years=assigned_years, - country_id=country_id, - category=category, - reference_category=reference_category_label, - ) - ) - - profile_diagnostics = pd.concat(profiles, ignore_index=True) - - return result, profile_diagnostics - - -def _impute_retired_dates_by_capacity_profile( - prepared_df: pd.DataFrame, - reference_capacity_df: pd.DataFrame, - lifetimes: dict[str, int], -) -> tuple[pd.Series, pd.Series, pd.DataFrame]: - """Impute dates for retired plants with neither date available.""" - imputed_start_year = pd.Series(np.nan, index=prepared_df.index, dtype=float) - imputed_end_year = pd.Series(np.nan, index=prepared_df.index, dtype=float) - profiles = [] - - missing_mask = ( - prepared_df["status"].eq(RETIRED) - & prepared_df["start_year"].isna() - & prepared_df["end_year"].isna() - ) - - if not missing_mask.any(): - return (imputed_start_year, imputed_end_year, pd.DataFrame()) - - grouped_missing = prepared_df.loc[missing_mask].groupby( - ["country_id", "category"], dropna=False - ) - - for (country_id, category), undated_group in grouped_missing: - reference_categories = _reference_categories(category) - reference_category_label = "+".join(reference_categories) - - group_mask = ( - prepared_df["country_id"].eq(country_id) - & prepared_df["category"].eq(category) - & prepared_df["status"].eq(RETIRED) - ) - - dated_group = prepared_df.loc[ - group_mask & prepared_df["end_year"].notna() - ].copy() - - reference_first_year = _reference_first_year( - reference_capacity_df=reference_capacity_df, - country_id=country_id, - categories=reference_categories, - max_year=_utils.DATASET_YEAR - 1, - ) - - observed_first_year = ( - int(dated_group["end_year"].min()) - if not dated_group.empty - else reference_first_year - ) - - first_profile_year = min(reference_first_year, observed_first_year) - - profile_years = pd.Index( - range(first_profile_year, _utils.DATASET_YEAR), name="end_year" - ) - - # Missing retirements are allocated only within the period covered - # by the reference capacity series. - allocatable_years = pd.Index( - range(reference_first_year, _utils.DATASET_YEAR), name="end_year" - ) - - reference_profile = _build_reference_retirement_profile( - reference_capacity_df=reference_capacity_df, - country_id=country_id, - categories=reference_categories, - years=profile_years, - ) - - reference_profile["profile_source"] = "negative_capacity_change" - - # Where the reference stock contains no reductions, use the timing of - # already dated retired plants as the retirement-profile basis. - if reference_profile["profile_fallback_used"].iloc[0]: - observed_retirement_basis = ( - dated_group.groupby("end_year")["output_capacity_mw"] - .sum() - .reindex(profile_years, fill_value=0.0) - ) - - if observed_retirement_basis.sum() > 0: - reference_profile["reference_profile_basis_mw"] = ( - observed_retirement_basis - ) - reference_profile["reference_profile_weight"] = ( - observed_retirement_basis / observed_retirement_basis.sum() - ) - reference_profile["profile_source"] = "observed_retirements" - - # Observed retirement dates may precede the reference series. - allocatable_years = profile_years - else: - reference_profile["profile_source"] = "uniform" - - missing_capacity_mw = undated_group["output_capacity_mw"].sum() - - allocation_profile = _build_clipped_residual_profile( - dated_df=dated_group, - profile=reference_profile, - allocatable_years=allocatable_years, - missing_capacity_mw=missing_capacity_mw, - year_col="end_year", - ) - - assigned_end_years = _allocate_years_by_target( - undated_df=undated_group, - target=allocation_profile["residual_target_mw"], - prefer_later_years=True, - ) - - assigned_start_years = assigned_end_years - undated_group["technology"].map( - lifetimes - ) - - imputed_end_year.loc[undated_group.index] = assigned_end_years - imputed_start_year.loc[undated_group.index] = assigned_start_years - - profiles.append( - _complete_retirement_profile( - profile=allocation_profile, - undated_df=undated_group, - assigned_end_years=assigned_end_years, - country_id=country_id, - category=category, - reference_category=reference_category_label, - ) - ) - - retirement_profiles = pd.concat(profiles, ignore_index=True) - - return (imputed_start_year, imputed_end_year, retirement_profiles) - - -def _impute_planned_start_years( - prepared_df: pd.DataFrame, - planned_commissioning_year_windows: dict[str, dict[str, list[int]]], -) -> tuple[pd.Series, pd.Series, pd.DataFrame]: - """Impute missing planned start years using flat capacity targets.""" - imputed_start_year = pd.Series(np.nan, index=prepared_df.index, dtype=float) - source_type = pd.Series(pd.NA, index=prepared_df.index, dtype="object") - profiles = [] - - missing_mask = ( - prepared_df["status"].isin(PLANNED) & prepared_df["start_year"].isna() - ) - - if not missing_mask.any(): - return (imputed_start_year, source_type, pd.DataFrame()) - - grouped_missing = prepared_df.loc[missing_mask].groupby( - ["country_id", "category", "technology", "status"], dropna=False - ) - - for (country_id, category, technology, status), undated_group in grouped_missing: - lower_offset, upper_offset = planned_commissioning_year_windows[technology][ - status - ] - - years = pd.Index( - range( - _utils.DATASET_YEAR + lower_offset, - _utils.DATASET_YEAR + upper_offset + 1, - ), - name="year", - ) - - missing_capacity_mw = undated_group["output_capacity_mw"].sum() - - flat_target = pd.Series( - missing_capacity_mw / len(years), - index=years, - dtype=float, - name="target_imputed_mw", - ) - - assigned_years = _allocate_years_by_target( - undated_df=undated_group, target=flat_target - ) - - imputed_start_year.loc[undated_group.index] = assigned_years - - source_type.loc[undated_group.index] = ( - f"imputed_{status.replace('-', '_')}_window" - ) - - profiles.append( - _complete_planned_commissioning_profile( - undated_df=undated_group, - assigned_years=assigned_years, - target=flat_target, - country_id=country_id, - category=category, - technology=technology, - status=status, - ) - ) - - profile_diagnostics = pd.concat(profiles, ignore_index=True) - - return (imputed_start_year, source_type, profile_diagnostics) - - -def _impute_remaining_start_years( - prepared_df: pd.DataFrame, - reference_capacity_df: pd.DataFrame, - lifetimes: dict[str, int], - method: str = "capacity_profile", -) -> tuple[pd.Series, pd.DataFrame]: - """Impute start years that remain missing after direct backfilling.""" - if method != "capacity_profile": - raise ValueError( - "Unknown start-year imputation method " - f"{method!r}. Expected 'capacity_profile'." - ) - - return _impute_start_years_by_capacity_profile( - prepared_df, reference_capacity_df=reference_capacity_df, lifetimes=lifetimes - ) - - -def _impute_start_year( - prepared_df: pd.DataFrame, - reference_capacity_df: pd.DataFrame, - lifetimes: dict[str, int], - planned_commissioning_year_windows: dict[str, dict[str, list[int]]], - method: str = "capacity_profile", -) -> tuple[pd.Series, pd.Series, pd.DataFrame, pd.DataFrame]: - """Impute missing powerplant start years and track source labels.""" - start_year = prepared_df["start_year"].copy() - start_year_source_type = _initial_year_source_type(start_year) - lifetime = prepared_df["technology"].map(lifetimes) - - # First, preserve the direct deterministic backfill from known end year. - direct_backfill_mask = start_year.isna() & prepared_df["end_year"].notna() - start_year.loc[direct_backfill_mask] = ( - prepared_df.loc[direct_backfill_mask, "end_year"] - - lifetime.loc[direct_backfill_mask] - ) - start_year_source_type.loc[direct_backfill_mask] = "derived_from_end_year" - - historical_profile_diagnostics = pd.DataFrame() - planned_profile_diagnostics = pd.DataFrame() - - # Impute planned projects independently of historical start-year - # methods. Only the missing planned capacity is distributed across - # the configured technology- and status-specific window. - planned_imputation_df = prepared_df.copy() - planned_imputation_df["start_year"] = start_year - - (planned_start_year, planned_source_type, planned_profile_diagnostics) = ( - _impute_planned_start_years( - prepared_df=planned_imputation_df, - planned_commissioning_year_windows=(planned_commissioning_year_windows), - ) - ) - - planned_imputed_mask = start_year.isna() & planned_start_year.notna() - - start_year.loc[planned_imputed_mask] = planned_start_year.loc[planned_imputed_mask] - start_year_source_type.loc[planned_imputed_mask] = planned_source_type.loc[ - planned_imputed_mask - ] - - # Apply the configured historical method only to historical plants - # that still have no start year. - historical_missing_mask = start_year.isna() & prepared_df["status"].isin(HISTORICAL) - - if historical_missing_mask.any(): - imputation_df = prepared_df.copy() - imputation_df["start_year"] = start_year - - (historical_start_year, historical_profile_diagnostics) = ( - _impute_remaining_start_years( - imputation_df, - reference_capacity_df=reference_capacity_df, - lifetimes=lifetimes, - method=method, - ) - ) - - historical_imputed_mask = ( - historical_missing_mask & historical_start_year.notna() - ) - - start_year.loc[historical_imputed_mask] = historical_start_year.loc[ - historical_imputed_mask - ] - start_year_source_type.loc[historical_imputed_mask] = f"imputed_{method}" - - return ( - start_year, - start_year_source_type, - historical_profile_diagnostics, - planned_profile_diagnostics, - ) - - -def _impute_end_year( - df: pd.DataFrame, lifetimes: dict[str, int], delay: dict[str, int] -) -> tuple[pd.Series, pd.Series]: - """Impute end_year using lifetime and track source labels. - - Old plants operating beyond lifetime will be retired with a given delay. - """ - ref_year = _utils.DATASET_YEAR - - end_year = df["end_year"].copy() - end_year_source_type = _initial_year_source_type(end_year) - - expected_end = df["start_year"] + df["technology"].map(lifetimes) - - lifetime_fill_mask = end_year.isna() & expected_end.notna() - result = end_year.copy() - result.loc[lifetime_fill_mask] = expected_end.loc[lifetime_fill_mask] - end_year_source_type.loc[lifetime_fill_mask] = "derived_from_start_year_lifetime" - - # Cap lifetime-derived end years so that plants recorded as retired - # end before the dataset year. - retired_lifetime_fill_mask = lifetime_fill_mask & df["status"].eq(RETIRED) - retired_end_capped_mask = retired_lifetime_fill_mask & result.ge(ref_year) - result.loc[retired_end_capped_mask] = ref_year - 1 - end_year_source_type.loc[retired_end_capped_mask] = ( - "derived_from_start_year_lifetime_capped_to_retired_status" - ) - - # Plants operating beyond expected lifetime will be retired after a delay - # of >=1 yr. - needs_delay = (result <= ref_year) & (df["status"].eq(OPERATING)) - delayed_end = result + df["technology"].map(delay).fillna(0).astype(int) - delayed_end = delayed_end.clip(lower=ref_year + 1) - - result.loc[needs_delay] = delayed_end.loc[needs_delay] - - end_year_source_type.loc[needs_delay & lifetime_fill_mask] = ( - "derived_from_start_year_lifetime_with_retirement_delay" - ) - - end_year_source_type.loc[needs_delay & ~lifetime_fill_mask] = ( - "observed_adjusted_with_retirement_delay" - ) - - return result, end_year_source_type - - -def _reconcile_status_from_observed_dates(df: pd.DataFrame) -> pd.Series: - """Correct status before imputation where observed dates are decisive. - - This function deliberately uses only the dates already present in the - input data. Observed retirement years are treated as the strongest signal: - if an observed end year is on or before the dataset year, the plant is - treated as retired before any lifetime or retirement-delay logic is - applied. - - Observed future start years are left in their original development stage - at this point. They are later collapsed to the final "planned" status after - complete start and end years have been assigned. - """ - ref_year = _utils.DATASET_YEAR - - corrected_status = df["status"].copy() - - observed_start_year = df["start_year"] - observed_end_year = df["end_year"] - - retired_from_observed_end = observed_end_year.notna() & observed_end_year.le( - ref_year - ) - corrected_status.loc[retired_from_observed_end] = RETIRED - - operating_from_observed_dates = ( - ~retired_from_observed_end - & observed_start_year.notna() - & observed_start_year.le(ref_year) - & observed_end_year.notna() - & observed_end_year.gt(ref_year) - ) - corrected_status.loc[operating_from_observed_dates] = OPERATING - - return corrected_status - - -def _impute_status(df: pd.DataFrame) -> pd.Series: - """Derive final temporal status after start/end years are complete.""" - status = df["status"].copy() - ref_year = _utils.DATASET_YEAR - status.loc[ref_year < df["start_year"]] = "planned" - status.loc[(df["start_year"] <= ref_year) & (ref_year < df["end_year"])] = OPERATING - status.loc[df["end_year"] <= ref_year] = RETIRED - - if status.isna().any(): - raise ValueError("Entries with ambiguous states were left in the dataframe.") - - return status - - -CAPACITY_DATE_EVENT_COLUMNS = [ - "powerplant_id", - "name", - "country_id", - "category", - "technology", - "status", - "year", - "event_type", - "source_type", - "source_label", - "output_capacity_mw", - "capacity_change_mw", -] - - -def _build_capacity_date_events(imputed: pd.DataFrame) -> pd.DataFrame: - """Convert imputed plant dates into annual commissioning and retirement events.""" - if imputed.empty: - return pd.DataFrame(columns=CAPACITY_DATE_EVENT_COLUMNS) - - common_cols = [ - "powerplant_id", - "name", - "country_id", - "category", - "technology", - "status", - "output_capacity_mw", - ] - - start_events = imputed[ - common_cols + ["start_year", "start_year_source_type"] - ].rename(columns={"start_year": "year", "start_year_source_type": "source_type"}) - start_events["event_type"] = "commissioning" - # Keep the plant capacity unchanged and create a signed event value for plotting: - # commissioning adds capacity, retirement removes capacity. This signed value is - # used only in the capacity-date-events diagnostic table. - start_events["capacity_change_mw"] = start_events["output_capacity_mw"] - - end_events = imputed[common_cols + ["end_year", "end_year_source_type"]].rename( - columns={"end_year": "year", "end_year_source_type": "source_type"} - ) - end_events["event_type"] = "retirement" - # Retirements are represented as negative events so they can be plotted below - # zero without altering the original plant capacity. - end_events["capacity_change_mw"] = -end_events["output_capacity_mw"] - - events = pd.concat([start_events, end_events], ignore_index=True) - - events["source_label"] = events["source_type"].map(_utils.date_source_labels()) - - return ( - events[CAPACITY_DATE_EVENT_COLUMNS] - .sort_values( - ["country_id", "year", "event_type", "source_type", "powerplant_id"] - ) - .reset_index(drop=True) - ) - - -def _get_time_imputation_colours() -> dict[str, tuple[float, float, float, float]]: - """Return colours for time-imputation source types.""" - start_year_sources = _utils.date_source_types_for("start_year") - end_year_sources = _utils.date_source_types_for("end_year") - - observed_sources = [ - source - for source in _utils.DATE_SOURCE_METADATA - if source in start_year_sources & end_year_sources - ] - start_only_sources = [ - source - for source in _utils.DATE_SOURCE_METADATA - if source in start_year_sources - end_year_sources - ] - end_only_sources = [ - source - for source in _utils.DATE_SOURCE_METADATA - if source in end_year_sources - start_year_sources - ] - - return ( - _plots.get_colour_dict( - observed_sources, "colorbrewer:Greys", value_range=(0.4, 0.45) - ) - | _plots.get_colour_dict( - start_only_sources, "colorbrewer:Purples", value_range=(0.2, 0.9) - ) - | _plots.get_colour_dict( - end_only_sources, "colorbrewer:Reds", value_range=(0.2, 0.9) - ) - ) - - -def plot_capacity_date_events( - events_df: pd.DataFrame, - commissioning_profile_df: pd.DataFrame, - retirement_profile_df: pd.DataFrame, - planned_profile_df: pd.DataFrame, - output_path: str, - cat: str, -) -> None: - """Plot commissioning and retirement events by date-source method.""" - display_category = cat.replace("_", " ") - suptitle = f"Capacity-date imputation for {display_category}" - - country_sets = [] - - for dataframe in [ - events_df, - commissioning_profile_df, - retirement_profile_df, - planned_profile_df, - ]: - if not dataframe.empty: - country_sets.extend(dataframe["country_id"].unique()) - - countries = sorted(set(country_sets)) - - if not countries: - _plots.plot_empty(suptitle, output_path) - return - - present_source_types = events_df["source_type"].unique().tolist() - - unknown_source_types = set(present_source_types) - set(_utils.DATE_SOURCE_METADATA) - - if unknown_source_types: - raise ValueError( - f"Missing DATE_SOURCE_METADATA entries for: {sorted(unknown_source_types)}" - ) - - source_types = [ - source_type - for source_type in _utils.DATE_SOURCE_METADATA - if source_type in present_source_types - ] - - source_colors = _get_time_imputation_colours() - - n_countries = len(countries) - cols = 2 if n_countries > 1 else 1 - rows = math.ceil(n_countries / cols) - - fig, axes = plt.subplots( - rows, - cols, - figsize=(cols * 7, rows * 4.5), - sharex=False, - sharey=False, - constrained_layout=True, - ) - axes_flat = np.array(axes).ravel() - - for ax, country in zip(axes_flat, countries): - country_events = events_df.loc[events_df["country_id"].eq(country)].copy() - - commissioning_target = ( - commissioning_profile_df.loc[ - commissioning_profile_df["country_id"].eq(country) - ].copy() - if not commissioning_profile_df.empty - else pd.DataFrame() - ) - - retirement_target = ( - retirement_profile_df.loc[ - retirement_profile_df["country_id"].eq(country) - ].copy() - if not retirement_profile_df.empty - else pd.DataFrame() - ) - - planned_target = ( - planned_profile_df.loc[planned_profile_df["country_id"].eq(country)].copy() - if not planned_profile_df.empty - else pd.DataFrame() - ) - - year_values = set(country_events["year"].astype(int)) - - if not commissioning_target.empty: - year_values.update(commissioning_target["year"].astype(int)) - - if not retirement_target.empty: - year_values.update(retirement_target["year"].astype(int)) - - if not planned_target.empty: - year_values.update(planned_target["year"].astype(int)) - - years = np.array(sorted(year_values)) - - if len(years) == 0: - _plots.draw_empty(ax, country, f"No date events for {country}") - continue - - country_events["year"] = country_events["year"].astype(int) - - annual_events = country_events.groupby( - ["year", "event_type", "source_type"], as_index=False - )["capacity_change_mw"].sum() - - positive_bottom = np.zeros(len(years)) - negative_bottom = np.zeros(len(years)) - - for source_type in source_types: - commissioning_values = ( - annual_events.loc[ - annual_events["source_type"].eq(source_type) - & annual_events["event_type"].eq("commissioning") - ] - .set_index("year")["capacity_change_mw"] - .reindex(years, fill_value=0.0) - .to_numpy() - ) - - retirement_values = ( - annual_events.loc[ - annual_events["source_type"].eq(source_type) - & annual_events["event_type"].eq("retirement") - ] - .set_index("year")["capacity_change_mw"] - .reindex(years, fill_value=0.0) - .to_numpy() - ) - - if commissioning_values.any(): - ax.bar( - years, - commissioning_values, - bottom=positive_bottom, - color=source_colors[source_type], - width=0.9, - ) - - if retirement_values.any(): - ax.bar( - years, - retirement_values, - bottom=negative_bottom, - color=source_colors[source_type], - width=0.9, - ) - - positive_bottom += commissioning_values - negative_bottom += retirement_values - - if not commissioning_target.empty: - commissioning_target = commissioning_target.sort_values("year") - - ax.plot( - commissioning_target["year"], - commissioning_target["target_final_mw"], - color="0.15", - linewidth=2, - ) - - if not retirement_target.empty: - retirement_target = retirement_target.sort_values("year") - - ax.plot( - retirement_target["year"], - -retirement_target["target_final_mw"], - color="0.15", - linewidth=2, - linestyle=":", - ) - - if not planned_target.empty: - planned_target = ( - planned_target.groupby("year", as_index=False)["target_imputed_mw"] - .sum() - .sort_values("year") - ) - - ax.plot( - planned_target["year"], - planned_target["target_imputed_mw"], - color="0.35", - linewidth=2, - linestyle="--", - ) - - ax.axhline(0, color="0.35", linewidth=0.8) - ax.set_title(country) - ax.set_xlabel("Year") - ax.set_ylabel("Annual capacity event (MW)") - ax.locator_params(axis="x", nbins=12) - ax.tick_params(axis="x", rotation=45) - ax.minorticks_off() - - for ax in axes_flat[n_countries:]: - ax.set_visible(False) - - legend_handles = [ - Patch( - facecolor=source_colors[source_type], - label=_utils.DATE_SOURCE_METADATA[source_type]["label"], - ) - for source_type in source_types - ] - - if not commissioning_profile_df.empty: - legend_handles.append( - Line2D( - [0], - [0], - color="0.15", - linewidth=2, - label="Commissioning profile for historic assets", - ) - ) - - if not retirement_profile_df.empty: - legend_handles.append( - Line2D( - [0], - [0], - color="0.15", - linewidth=2, - linestyle=":", - label="Retirement profile for historic assets", - ) - ) - - if not planned_profile_df.empty: - legend_handles.append( - Line2D( - [0], - [0], - color="0.35", - linewidth=2, - linestyle="--", - label="Commissioning profile for planned assets", - ) - ) - - fig.legend( - handles=legend_handles, - loc="center left", - bbox_to_anchor=(1.0, 0.5), - frameon=False, - ) - fig.suptitle(suptitle, fontsize=14) - - fig.savefig(output_path, bbox_inches="tight") - plt.close(fig) - - -def impute( - relocated_gdf: gpd.GeoDataFrame, - reference_capacity_df: pd.DataFrame, - imputation: dict, - technology_mapping: dict, -) -> tuple[gpd.GeoDataFrame, pd.DataFrame, pd.DataFrame, pd.DataFrame]: - """Add automatic and user imputations to fill missing data. - - Args: - relocated_gdf: Relocated powerplants with country identifiers. - reference_capacity_df: Annual category-level capacity stock data. - imputation: Imputation configuration. - technology_mapping: Technology-mapping configuration. - """ - commissioning_profile_df = pd.DataFrame() - retirement_profile_df = pd.DataFrame() - planned_profile_df = pd.DataFrame() - - _utils.check_single_category(relocated_gdf) - - lifetimes = imputation["lifetime_years"] - planned_commissioning_year_windows = imputation[ - "planned_commissioning_year_windows" - ] - retirement_delay_years = imputation["retirement_delay_years"] - scenario = SCENARIO_MAP[imputation["scenario"]] - start_year_imputation_method = imputation["start_year_imputation_method"] - - status_after_observed_date_correction = _reconcile_status_from_observed_dates( - relocated_gdf - ) - - # Get facilities within the requested scenario after correcting only - # the statuses that are already contradicted by observed dates. - scenario_gdf = relocated_gdf.copy() - scenario_gdf["status"] = status_after_observed_date_correction - - imputed = scenario_gdf[scenario_gdf["status"].isin(scenario)].copy() - - if not imputed.empty: - ( - imputed["start_year"], - start_year_source_type, - commissioning_profile_df, - planned_profile_df, - ) = _impute_start_year( - prepared_df=imputed, - reference_capacity_df=reference_capacity_df, - lifetimes=lifetimes, - planned_commissioning_year_windows=(planned_commissioning_year_windows), - method=start_year_imputation_method, - ) - - imputed["end_year"], end_year_source_type = _impute_end_year( - imputed, lifetimes, retirement_delay_years - ) - - retired_start_year, retired_end_year, retirement_profile_df = ( - _impute_retired_dates_by_capacity_profile( - prepared_df=imputed, - reference_capacity_df=reference_capacity_df, - lifetimes=lifetimes, - ) - ) - - retirement_imputed_mask = retired_end_year.notna() - - imputed.loc[retirement_imputed_mask, "start_year"] = retired_start_year.loc[ - retirement_imputed_mask - ] - - imputed.loc[retirement_imputed_mask, "end_year"] = retired_end_year.loc[ - retirement_imputed_mask - ] - - start_year_source_type.loc[retirement_imputed_mask] = ( - "derived_from_imputed_retirement_end_year" - ) - end_year_source_type.loc[retirement_imputed_mask] = ( - "imputed_retirement_capacity_profile" - ) - - has_complete_dates = imputed[["start_year", "end_year"]].notna().all(axis=1) - - status_final = pd.Series(pd.NA, index=imputed.index, dtype="object") - - if has_complete_dates.any(): - status_final.loc[has_complete_dates] = _impute_status( - imputed.loc[has_complete_dates] - ) - - # Drop projects with insufficient date data. - imputed = imputed.loc[has_complete_dates].copy() - - # Update the powerplant status. - imputed["status"] = status_final.loc[imputed.index] - - # Pass date-source provenance forward. - imputed["start_year_source_type"] = start_year_source_type.loc[imputed.index] - imputed["end_year_source_type"] = end_year_source_type.loc[imputed.index] - else: - imputed["start_year_source_type"] = pd.Series(dtype="object") - imputed["end_year_source_type"] = pd.Series(dtype="object") - - schema = _schemas.build_schema(technology_mapping, "impute") - return ( - schema.validate(imputed), - commissioning_profile_df, - retirement_profile_df, - planned_profile_df, - ) - - -def explore(imputed: gpd.GeoDataFrame, output_path: str, colormap="tab20"): - """Create a HTML map for users to explore.""" - if imputed.empty: - with open(output_path, "w") as f: - f.write("No data") - else: - explorer = imputed.explore( - column="technology", legend=True, popup=True, cmap=colormap - ) - explorer.save(output_path) - - -def plot_powerplant_capacity_buildup( - df: pd.DataFrame, output_path: str, colormap: str, cat: str = "powerplant" -): - """Plot stacked bar charts of active powerplant capacity over time per country. - - Input should be a powerplant capacity file of a single category. - """ - suptitle = f"Active {cat} capacity by technology per country" - - if df.empty: - _plots.plot_empty(suptitle, output_path) - return - - # Year range (x-axis) - start_year = df["start_year"].astype(int).min() - end_year = df["end_year"].astype(int).max() - years = list(range(start_year, end_year + 1)) - - # Layout (per country in alphabetical order) - countries = sorted(df["country_id"].unique()) - n_countries = len(countries) - cols = 2 if n_countries > 1 else 1 - rows = math.ceil(n_countries / cols) - - # Tech type color range - tech_types = sorted(df["technology"].unique()) - cmap = Colormap(colormap).to_mpl() - colors = [cmap(i) for i in np.linspace(0, 1, len(tech_types))] - - # Figure (always 2 columns, flexible rows) - fig, axes = plt.subplots( - rows, - cols, - figsize=(cols * 5, rows * 4), - sharex=False, - sharey=False, - constrained_layout=True, - ) - axes_flat = np.array(axes).ravel() - - # Plot per country - for ax, country in zip(axes_flat, countries): - country_df = df[df["country_id"] == country] - if country_df.empty: - _plots.draw_empty(ax, country, f"No data for {country}") - continue - - cap_mw = pd.DataFrame(0.0, index=years, columns=tech_types) - for year in years: - active = country_df[ - (country_df["start_year"] <= year) & (year < country_df["end_year"]) - ] - cap_mw.loc[year] = ( - active.groupby("technology")["output_capacity_mw"] - .sum() - .reindex(tech_types, fill_value=0) - ) - - cap_mw.plot(kind="bar", stacked=True, ax=ax, color=colors, legend=False, rot=45) - ax.set_title(country) - ax.set_ylabel("Capacity (MW)") - ax.locator_params(axis="x", nbins=10) - ax.minorticks_off() - - # Hide extra axes - for ax in axes_flat[n_countries:]: - ax.set_visible(False) - - # Add details - handles, labels = axes_flat[0].get_legend_handles_labels() - fig.legend( - handles[::-1], - labels[::-1], - loc="center left", - bbox_to_anchor=(1.0, 0.5), - title="Technology", - frameon=False, - ) - fig.suptitle(suptitle, fontsize=14) - - fig.savefig(output_path, bbox_inches="tight") - - -def main() -> None: - """Main snakemake process.""" - relocated_gdf = gpd.read_parquet(snakemake.input.relocated) - reference_capacity_df = pd.read_parquet(snakemake.input.category_capacity) - if relocated_gdf.empty: - imputed_gdf = relocated_gdf - imputed_gdf["start_year_source_type"] = pd.Series(dtype="object") - imputed_gdf["end_year_source_type"] = pd.Series(dtype="object") - commissioning_profile_df = pd.DataFrame() - retirement_profile_df = pd.DataFrame() - planned_profile_df = pd.DataFrame() - else: - ( - imputed_gdf, - commissioning_profile_df, - retirement_profile_df, - planned_profile_df, - ) = impute( - relocated_gdf=relocated_gdf, - reference_capacity_df=reference_capacity_df, - imputation=snakemake.params.imputation, - technology_mapping=snakemake.params.tech_map, - ) - imputed_gdf.to_parquet(snakemake.output.aged) - - capacity_date_events_df = _build_capacity_date_events(imputed_gdf) - - capacity_date_events_df.to_parquet( - snakemake.output.capacity_date_events, index=False - ) - - plot_powerplant_capacity_buildup( - imputed_gdf, - snakemake.output.histogram, - "seaborn:tab20", - snakemake.wildcards.category, - ) - explore(imputed_gdf, snakemake.output.explorer) - plot_capacity_date_events( - events_df=capacity_date_events_df, - commissioning_profile_df=commissioning_profile_df, - retirement_profile_df=retirement_profile_df, - planned_profile_df=planned_profile_df, - output_path=snakemake.output.capacity_date_plot, - cat=snakemake.wildcards.category, - ) - - -if __name__ == "__main__": - sys.stderr = open(snakemake.log[0], "w", buffering=1) - main() diff --git a/workflow/scripts/impute_time.py b/workflow/scripts/impute_time.py new file mode 100644 index 0000000..e84589e --- /dev/null +++ b/workflow/scripts/impute_time.py @@ -0,0 +1,1081 @@ +"""Impute powerplant dates and produce time-imputation diagnostics.""" + +import heapq +import math +import sys +from collections.abc import Mapping +from typing import TYPE_CHECKING, Any, Literal + +import _plots +import _schemas +import _utils +import geopandas as gpd +import numpy as np +import pandas as pd +from _schemas import HISTORICAL, OPERATING, PLANNED, RETIRED, SCENARIO_MAP +from _utils import DATASET_YEAR, EIA_CAT_MAPPING +from matplotlib import pyplot as plt +from matplotlib.lines import Line2D +from matplotlib.patches import Patch +from matplotlib.typing import ColorType + +if TYPE_CHECKING: + snakemake: Any + + +PROFILE_COLUMNS = [ + "country_id", + "category", + "reference_category", + "event_type", + "cohort", + "technology", + "status", + "profile_source", + "year", + "target_mw", + "observed_mw", + "imputed_mw", +] + + +def empty_profiles() -> pd.DataFrame: + """Return an empty normalized profile table.""" + return pd.DataFrame(columns=PROFILE_COLUMNS) + + +def _initialise_year_source(year: pd.Series) -> pd.Series: + """Label whether year values were originally observed or missing.""" + source_type = pd.Series("observed", index=year.index, dtype="object") + source_type.loc[year.isna()] = "missing_unresolved" + return source_type + + +def _has_no_dates(plants: pd.DataFrame) -> pd.Series: + """Return powerplants without an observed start or end year.""" + return plants[["start_year", "end_year"]].isna().all(axis=1) + + +def _reference_capacity_stock( + reference_capacity_df: pd.DataFrame, + country_id: str, + categories: list[str], + max_year: int, +) -> pd.Series: + """Return annual reference stock summed across mapped categories. + + One powerplant category may map to multiple reference categories. For + example, hydropower combines the hydropower and pumped-storage stocks. + These are summed by year before deriving event profiles. + """ + return ( + reference_capacity_df.loc[ + reference_capacity_df["country_id"].eq(country_id) + & reference_capacity_df["category"].isin(categories) + & reference_capacity_df["year"].le(max_year), + ["year", "capacity_mw"], + ] + .groupby("year")["capacity_mw"] + .sum(min_count=1) + .sort_index() + ) + + +def _first_reported_year(capacity_stock: pd.Series) -> int | None: + """Return the first reported stock year, if one exists.""" + reported_years = capacity_stock.index[capacity_stock.notna()] + return None if reported_years.empty else int(reported_years.min()) + + +def _build_reference_profile( + reference_capacity_df: pd.DataFrame, + country_id: str, + categories: list[str], + years: pd.Index, + event_type: Literal["commissioning", "retirement"], + *, + smoothing_window: int = 1, +) -> pd.DataFrame: + """Build a commissioning or retirement profile from annual stock changes.""" + capacity_stock = _reference_capacity_stock( + reference_capacity_df, country_id, categories, years.max() + ) + + # Annual stock changes are only a proxy for gross events. + # - Positive net changes provide commissioning weight + # - Negative net changes provide retirement weight. + # Changes in the opposite direction are clipped out. + direction = 1 if event_type == "commissioning" else -1 + change = (direction * capacity_stock.diff()).clip(lower=0).reindex(years).fillna(0) + basis = change.copy() + if smoothing_window > 1: + basis = basis.rolling(smoothing_window, center=True, min_periods=1).mean() + + fallback_used = basis.sum() <= 0 + if fallback_used: + # A uniform 'flat' profile keeps dates imputable when reference history is + # absent, or contains no changes in the requested direction. + basis.loc[:] = 1.0 + + return pd.DataFrame( + { + "reference_stock_mw": capacity_stock.reindex(years), + "reference_change_mw": change, + "reference_profile_basis_mw": basis, + "reference_profile_weight": basis / basis.sum(), + "profile_fallback_used": fallback_used, + }, + index=years, + ) + + +def _build_residual_profile( + dated_df: pd.DataFrame, + profile: pd.DataFrame, + allocatable_years: pd.Index, + missing_capacity_mw: float, + year_col: str, +) -> pd.DataFrame: + """Scale a reference profile and subtract capacity with observed dates.""" + result = profile.copy() + observed = ( + dated_df.groupby(year_col)["output_capacity_mw"] + .sum() + .reindex(result.index, fill_value=0.0) + ) + + # Reference data determine the temporal shape, while the plant dataset + # determines the total capacity represented by the scaled target. + total_capacity_mw = observed.sum() + missing_capacity_mw + result["observed_mw"] = observed + result["target_final_mw"] = result["reference_profile_weight"] * total_capacity_mw + + # Exclude observed overcapacity from the initial residual target. If no + # positive residual remains, fall back to the reference or uniform profile. + residual = ( + (result["target_final_mw"] - observed) + .clip(lower=0) + .reindex(allocatable_years, fill_value=0.0) + ) + if residual.sum() <= 0: + residual = result["reference_profile_weight"].reindex( + allocatable_years, fill_value=0.0 + ) + if residual.sum() <= 0: + residual = pd.Series(1.0, index=allocatable_years) + + # Rescale the feasible residual so every missing capacity block is + # allocated while preserving the residual profile's relative shape. + residual = residual / residual.sum() * missing_capacity_mw + result["residual_target_mw"] = residual.reindex(result.index, fill_value=0.0) + return result + + +def _allocate_years_by_target( + plants: pd.DataFrame, + target: pd.Series, + *, + earliest_year: pd.Series | None = None, + prefer_later_years: bool = False, +) -> pd.Series: + """Greedily allocate whole plants using an incremental priority queue. + + Earlier/later mean lower/higher calendar-year values, respectively. + """ + if target.empty: + raise ValueError("Cannot allocate dates against an empty target profile.") + if earliest_year is None: + earliest_year = pd.Series(target.index.min(), index=plants.index) + + # Allocate the least flexible plants first, then the largest capacity + # blocks, to reduce poor fits caused by indivisible plants. + order = pd.DataFrame( + { + "earliest_year": earliest_year, + "capacity": plants["output_capacity_mw"], + "powerplant_id": plants["powerplant_id"], + "row_order": np.arange(len(plants)), + }, + index=plants.index, + ).sort_values( + ["earliest_year", "capacity", "powerplant_id", "row_order"], + ascending=[False, False, True, True], + ) + + remaining = target.copy() + years_desc = sorted(target.index, reverse=True) + next_year = 0 + heap: list[tuple[float, float, int, int]] = [] + assigned = pd.Series(np.nan, index=plants.index, dtype=float) + + def push(year: int) -> None: + """Add a year to the allocation queue with its current priority.""" + year_tie = -year if prefer_later_years else year + # heapq uses a min-heap algorithm + # negating values prioritize the largest deficit and then the largest target. + heapq.heappush(heap, (-remaining.loc[year], -target.loc[year], year_tie, year)) + + for plant_index in order.index: + lower_bound = earliest_year.loc[plant_index] + + # Plants are ordered by decreasing earliest year + # As next_year increments, so newly feasible plants are added to the queue + while next_year < len(years_desc) and years_desc[next_year] >= lower_bound: + push(years_desc[next_year]) + next_year += 1 + + if not heap: + raise ValueError(f"No allocatable year at or after {lower_bound}.") + + # Heap priority is deterministic: + # largest remaining deficit, then original target, then the configured year tie. + _, _, _, assigned_year = heapq.heappop(heap) + assigned.loc[plant_index] = assigned_year + remaining.loc[assigned_year] -= plants.loc[plant_index, "output_capacity_mw"] + # Reinsert the year with its reduced remaining capacity. + push(assigned_year) + + return assigned + + +def _profile_rows( + profile: pd.DataFrame, + plants: pd.DataFrame, + assigned_years: pd.Series, + *, + country_id: str, + category: str, + event_type: str, + cohort: str, + profile_source: str, + reference_category: str | None = None, + technology: str | None = None, + status: str | None = None, +) -> pd.DataFrame: + """Return normalized diagnostic rows for an imputed plant cohort.""" + imputed = ( + pd.DataFrame( + {"year": assigned_years, "output_capacity_mw": plants["output_capacity_mw"]} + ) + .groupby("year")["output_capacity_mw"] + .sum() + .reindex(profile.index, fill_value=0.0) + ) + return pd.DataFrame( + { + "country_id": country_id, + "category": category, + "reference_category": reference_category, + "event_type": event_type, + "cohort": cohort, + "technology": technology, + "status": status, + "profile_source": profile_source, + "year": profile.index, + "target_mw": profile["target_final_mw"], + "observed_mw": profile["observed_mw"], + "imputed_mw": imputed, + } + )[PROFILE_COLUMNS] + + +def _remaining_target( + target: pd.Series, plants: pd.DataFrame, assigned_years: pd.Series +) -> pd.Series: + """Subtract assigned plant capacity from an annual target.""" + assigned_capacity = ( + plants.assign(year=assigned_years) + .groupby("year")["output_capacity_mw"] + .sum() + .reindex(target.index, fill_value=0.0) + ) + return target - assigned_capacity + + +def _allocate_historical_start_years( + plants: pd.DataFrame, target: pd.Series, earliest_plant_start_years: pd.Series +) -> pd.Series: + """Allocate retired and operating starts against one commissioning target.""" + assigned = pd.Series(np.nan, index=plants.index, dtype=float) + + # A retired plant needs at least one operating year before its end year. + retired = plants.loc[plants["status"].eq(RETIRED)] + if not retired.empty: + latest_retired_start_year = DATASET_YEAR - 1 + retired_years = _allocate_years_by_target( + retired, target.loc[target.index <= latest_retired_start_year] + ) + assigned.loc[retired.index] = retired_years + target = _remaining_target(target, retired, retired_years) + + # Operating plants must be commissioned late enough to remain active in the + # dataset year under their configured lifetime assumption. + if not earliest_plant_start_years.empty: + operating = plants.loc[earliest_plant_start_years.index] + assigned.loc[operating.index] = _allocate_years_by_target( + operating, target, earliest_year=earliest_plant_start_years + ) + + return assigned + + +def _impute_historical_start_years( + plants: pd.DataFrame, + reference_capacity_df: pd.DataFrame, + lifetimes: Mapping[str, int], +) -> tuple[pd.Series, pd.DataFrame]: + """Impute historical start years from commissioning profiles.""" + result = pd.Series(np.nan, index=plants.index, dtype=float) + profiles: list[pd.DataFrame] = [] + missing = _has_no_dates(plants) & plants["status"].isin(HISTORICAL) + + for (country_id, category), undated in plants.loc[missing].groupby( + ["country_id", "category"] + ): + # Harmonise plant categories with the annual reference-capacity categories. + categories = EIA_CAT_MAPPING[category] + group = plants["country_id"].eq(country_id) & plants["category"].eq(category) + dated = plants.loc[ + group & plants["status"].isin(HISTORICAL) & plants["start_year"].notna() + ] + + operating = undated["status"].eq(OPERATING) + lifetime = undated["technology"].map(lifetimes) + earliest_operating_year = DATASET_YEAR - lifetime.loc[operating] + 1 + retired = undated["status"].eq(RETIRED) + + stock = _reference_capacity_stock( + reference_capacity_df, country_id, categories, DATASET_YEAR + ) + reference_first = _first_reported_year(stock) + first_year = min( + pd.concat( + [ + earliest_operating_year, + DATASET_YEAR - lifetime.loc[retired], + dated["start_year"], + ] + ).min(), + reference_first if reference_first is not None else DATASET_YEAR, + ) + years = pd.Index(range(int(first_year), DATASET_YEAR + 1), name="start_year") + last_allocatable_year = DATASET_YEAR if operating.any() else DATASET_YEAR - 1 + allocatable_years = years[years <= last_allocatable_year] + + profile = _build_reference_profile( + reference_capacity_df, country_id, categories, years, "commissioning" + ) + allocation = _build_residual_profile( + dated, + profile, + allocatable_years, + undated["output_capacity_mw"].sum(), + "start_year", + ) + assigned = _allocate_historical_start_years( + undated, allocation["residual_target_mw"], earliest_operating_year + ) + result.loc[undated.index] = assigned + source = ( + "uniform" + if bool(profile["profile_fallback_used"].iloc[0]) + else "positive_capacity_change" + ) + profiles.append( + _profile_rows( + allocation, + undated, + assigned, + country_id=country_id, + category=category, + event_type="commissioning", + cohort="historical", + profile_source=source, + reference_category="+".join(categories), + ) + ) + + return result, pd.concat( + profiles, ignore_index=True + ) if profiles else empty_profiles() + + +def _impute_retired_end_years( + plants: pd.DataFrame, + retirement_linked: pd.Series, + reference_capacity_df: pd.DataFrame, +) -> tuple[pd.Series, pd.DataFrame]: + """Impute retirement years for plants whose two dates were missing.""" + end_result = pd.Series(np.nan, index=plants.index, dtype=float) + profiles: list[pd.DataFrame] = [] + + for (country_id, category), undated in plants.loc[retirement_linked].groupby( + ["country_id", "category"] + ): + categories = EIA_CAT_MAPPING[category] + group = plants["country_id"].eq(country_id) & plants["category"].eq(category) + dated = plants.loc[ + group & plants["status"].eq(RETIRED) & plants["end_year"].notna() + ] + stock = _reference_capacity_stock( + reference_capacity_df, country_id, categories, DATASET_YEAR + ) + reference_first = _first_reported_year(stock) + observed_first = int(dated["end_year"].min()) if not dated.empty else None + earliest_end = undated["start_year"] + 1 + first_year = int( + min( + year + for year in (reference_first, observed_first, earliest_end.min()) + if year is not None + ) + ) + years = pd.Index(range(first_year, DATASET_YEAR + 1), name="end_year") + profile = _build_reference_profile( + reference_capacity_df, country_id, categories, years, "retirement" + ) + # Missing retirements are normally restricted to the period covered by + # the country's reference series. + allocatable_years = pd.Index( + range(reference_first or first_year, DATASET_YEAR + 1), name="end_year" + ) + profile_source = "negative_capacity_change" + + if bool(profile["profile_fallback_used"].iloc[0]): + # When stock history contains no reductions, observed retirement + # timing is more informative than a uniform fallback. It may also + # legitimately precede the reference-series coverage. + observed_basis = ( + dated.groupby("end_year")["output_capacity_mw"] + .sum() + .reindex(years, fill_value=0.0) + ) + if observed_basis.sum() > 0: + profile["reference_profile_weight"] = ( + observed_basis / observed_basis.sum() + ) + profile_source = "observed_retirements" + allocatable_years = years + else: + profile_source = "uniform" + + allocation = _build_residual_profile( + dated, + profile, + allocatable_years, + undated["output_capacity_mw"].sum(), + "end_year", + ) + assigned_end = _allocate_years_by_target( + undated, + allocation["residual_target_mw"], + earliest_year=earliest_end, + prefer_later_years=True, + ) + end_result.loc[undated.index] = assigned_end + profiles.append( + _profile_rows( + allocation, + undated, + assigned_end, + country_id=country_id, + category=category, + event_type="retirement", + cohort="historical", + profile_source=profile_source, + reference_category="+".join(categories), + ) + ) + + combined = pd.concat(profiles, ignore_index=True) if profiles else empty_profiles() + return end_result, combined + + +def _impute_planned_from_commissioning_windows( + plants: pd.DataFrame, windows: Mapping[str, Mapping[str, list[int]]] +) -> tuple[pd.DataFrame, pd.DataFrame]: + """Assign start years to fully undated planned plants using flat windows.""" + result = plants.copy() + profiles: list[pd.DataFrame] = [] + missing = result["status"].isin(PLANNED) & _has_no_dates(result) + + for (country_id, category, technology, status), undated in result.loc[ + missing + ].groupby(["country_id", "category", "technology", "status"]): + # Planned projects follow technology-specific future windows + lower, upper = windows[technology][status] + years = pd.Index( + range(DATASET_YEAR + lower, DATASET_YEAR + upper + 1), name="year" + ) + target = pd.Series( + undated["output_capacity_mw"].sum() / len(years), index=years + ) + assigned = _allocate_years_by_target(undated, target) + result.loc[undated.index, "start_year"] = assigned + result.loc[undated.index, "start_year_source_type"] = ( + f"imputed_{status.replace('-', '_')}_window" + ) + + profile = pd.DataFrame( + {"target_final_mw": target, "observed_mw": 0.0}, index=years + ) + profiles.append( + _profile_rows( + profile, + undated, + assigned, + country_id=country_id, + category=category, + event_type="commissioning", + cohort="planned", + profile_source="uniform_window", + technology=technology, + status=status, + ) + ) + + combined = pd.concat(profiles, ignore_index=True) if profiles else empty_profiles() + return result, combined + + +def _impute_years_from_lifetimes( + plants: pd.DataFrame, lifetimes: Mapping[str, int], delays: Mapping[str, int] +) -> pd.DataFrame: + """Complete missing start/end year counterparts using configured lifetimes.""" + result = plants.copy() + lifetime = result["technology"].map(lifetimes) + + expected_start = result["end_year"] - lifetime + start_mask = result["start_year"].isna() & expected_start.notna() + result.loc[start_mask, "start_year"] = expected_start.loc[start_mask] + result.loc[start_mask, "start_year_source_type"] = "derived_from_end_year" + + expected_end = result["start_year"] + lifetime + end_mask = result["end_year"].isna() & expected_end.notna() + result.loc[end_mask, "end_year"] = expected_end.loc[end_mask] + result.loc[end_mask, "end_year_source_type"] = "derived_from_start_year_lifetime" + + # Only derived dates may be adjusted: known start/end years remain sacred. + # A plant marked retired must be offline by the start of the dataset year. + retired_cap = ( + end_mask & result["status"].eq(RETIRED) & result["end_year"].gt(DATASET_YEAR) + ) + result.loc[retired_cap, "end_year"] = DATASET_YEAR + result.loc[retired_cap, "end_year_source_type"] = ( + "derived_from_start_year_lifetime_capped_to_retired_status" + ) + + # An operating plant that outlived its assumed lifetime receives the + # configured extension, with at least one year beyond the dataset year. + delayed = ( + end_mask & result["status"].eq(OPERATING) & result["end_year"].le(DATASET_YEAR) + ) + result.loc[delayed, "end_year"] = ( + result.loc[delayed, "end_year"] + result.loc[delayed, "technology"].map(delays) + ).clip(lower=DATASET_YEAR + 1) + result.loc[delayed, "end_year_source_type"] = ( + "derived_from_start_year_lifetime_with_retirement_delay" + ) + return result + + +def _impute_capacity_profile_dates( + plants: pd.DataFrame, + reference_capacity_df: pd.DataFrame, + lifetimes: Mapping[str, int], + delays: Mapping[str, int], +) -> tuple[pd.DataFrame, pd.DataFrame]: + """Complete fully undated historical plants using capacity profiles.""" + result = plants.copy() + undated = _has_no_dates(result) & result["status"].isin(HISTORICAL) + retired = undated & result["status"].eq(RETIRED) + + start_year, commissioning_profiles = _impute_historical_start_years( + result, reference_capacity_df, lifetimes + ) + start_imputed = undated & start_year.notna() + result.loc[start_imputed, "start_year"] = start_year.loc[start_imputed] + result.loc[start_imputed, "start_year_source_type"] = "imputed_capacity_profile" + result.loc[retired & start_year.notna(), "start_year_source_type"] = ( + "imputed_capacity_profile_retirement_linked" + ) + + # Profile-imputed commissioning years set the earliest feasible retirement year. + end_year, retirement_profiles = _impute_retired_end_years( + result, retired, reference_capacity_df + ) + end_imputed = retired & end_year.notna() + result.loc[end_imputed, "end_year"] = end_year.loc[end_imputed] + result.loc[end_imputed, "end_year_source_type"] = ( + "imputed_retirement_capacity_profile" + ) + + # Complete counterparts which the capacity profiles left unresolved with lifetimes. + completed = _impute_years_from_lifetimes(result.loc[undated], lifetimes, delays) + date_columns = [ + "start_year", + "end_year", + "start_year_source_type", + "end_year_source_type", + ] + result.loc[undated, date_columns] = completed[date_columns] + + profile_parts = [ + profile + for profile in (commissioning_profiles, retirement_profiles) + if not profile.empty + ] + profiles = ( + pd.concat(profile_parts, ignore_index=True).reindex(columns=PROFILE_COLUMNS) + if profile_parts + else empty_profiles() + ) + return result, profiles + + +def _adjust_scenario_status_to_reference_year(plants: pd.DataFrame) -> pd.Series: + """Adjust powerplant status to the given year without altering observed dates.""" + status = plants["status"].copy() + + # Observed retirement years are the strongest status signal. + retired = plants["end_year"].notna() & plants["end_year"].le(DATASET_YEAR) + status.loc[retired] = RETIRED + + # Ensure powerplants active in the given year are marked as operating + operating = ( + ~retired + & plants["start_year"].notna() + & plants["start_year"].le(DATASET_YEAR) + & plants["end_year"].notna() + & plants["end_year"].gt(DATASET_YEAR) + ) + status.loc[operating] = OPERATING + + # Find powerplants known to be operating after the provided year. + # Treat them as in construction for scenario eligibility. + future_historical = ( + ~retired + & plants["start_year"].notna() + & plants["start_year"].gt(DATASET_YEAR) + & status.isin(HISTORICAL) + ) + status.loc[future_historical] = "construction" + return status + + +def _simplify_status_for_users(plants: pd.DataFrame) -> pd.Series: + """Derive final temporal status from completed start and end years.""" + status = plants["status"].copy() + status.loc[plants["start_year"].gt(DATASET_YEAR)] = "planned" + status.loc[ + plants["start_year"].le(DATASET_YEAR) & plants["end_year"].gt(DATASET_YEAR) + ] = OPERATING + status.loc[plants["end_year"].le(DATASET_YEAR)] = RETIRED + return status + + +def impute_time( + plants: pd.DataFrame, reference_capacity_df: pd.DataFrame, imputation: Mapping +) -> tuple[pd.DataFrame, pd.DataFrame]: + """Impute missing dates and return plants plus normalized profile diagnostics.""" + if plants.empty: + return _schemas.PlantSchema.empty(), empty_profiles() + + _utils.check_single_category(plants) + + scenario_plants = plants.assign( + status=_adjust_scenario_status_to_reference_year(plants), + start_year_source_type=_initialise_year_source(plants["start_year"]), + end_year_source_type=_initialise_year_source(plants["end_year"]), + ) + result = scenario_plants.loc[ + scenario_plants["status"].isin(SCENARIO_MAP[imputation["scenario"]]) + ].copy() + + if result.empty: + # Scenario removed all available powerplants + return _schemas.PlantSchema.empty(), empty_profiles() + + # Observed dates should be immutable! + # Start by filling in start/end years that have a known counterpart. + lifetimes = imputation["lifetime_years"] + delays = imputation["retirement_delay_years"] + result = _impute_years_from_lifetimes(result, lifetimes, delays) + + # Fill unknown planned plants using flat commissioning windows. + # Then fill end years with the lifetime. + result, planned_profiles = _impute_planned_from_commissioning_windows( + plants=result, windows=imputation["planned_commissioning_year_windows"] + ) + result = _impute_years_from_lifetimes(result, lifetimes, delays) + + # Remaining historical powerplants are filled with user-specified methods. + match imputation["method"]: + case "capacity_profile": + result, method_profiles = _impute_capacity_profile_dates( + result, reference_capacity_df, lifetimes, delays + ) + case method: + raise ValueError(f"Unknown time imputation method: {method}") + + result["status"] = _simplify_status_for_users(result) + profile_parts = [i for i in (planned_profiles, method_profiles) if not i.empty] + profiles = ( + pd.concat(profile_parts, ignore_index=True).reindex(columns=PROFILE_COLUMNS) + if profile_parts + else empty_profiles() + ) + return result, profiles + + +CAPACITY_DATE_EVENT_COLUMNS = [ + "powerplant_id", + "name", + "country_id", + "category", + "technology", + "status", + "year", + "event_type", + "source_type", + "source_label", + "output_capacity_mw", + "capacity_change_mw", +] + + +def _build_capacity_date_events(imputed: pd.DataFrame) -> pd.DataFrame: + """Convert imputed plant dates into annual commissioning and retirement events.""" + common = [ + "powerplant_id", + "name", + "country_id", + "category", + "technology", + "status", + "output_capacity_mw", + ] + events = [] + for year_col, source_col, event_type, sign in ( + ("start_year", "start_year_source_type", "commissioning", 1), + ("end_year", "end_year_source_type", "retirement", -1), + ): + event = imputed[common + [year_col, source_col]].rename( + columns={year_col: "year", source_col: "source_type"} + ) + event["event_type"] = event_type + + # Keep plant capacity unchanged and add a signed event value solely for + # diagnostics: commissioning is positive and retirement is negative. + event["capacity_change_mw"] = sign * event["output_capacity_mw"] + events.append(event) + + result = pd.concat(events, ignore_index=True) + result["source_label"] = result["source_type"].map(_utils.date_source_labels()) + return ( + result[CAPACITY_DATE_EVENT_COLUMNS] + .sort_values( + ["country_id", "year", "event_type", "source_type", "powerplant_id"] + ) + .reset_index(drop=True) + ) + + +def _time_imputation_colours() -> dict[str, ColorType]: + """Return colours for time imputation source types.""" + start_sources = _utils.date_source_types_for("start_year") + end_sources = _utils.date_source_types_for("end_year") + ordered = _utils.DATE_SOURCE_METADATA + shared = [source for source in ordered if source in start_sources & end_sources] + start_only = [source for source in ordered if source in start_sources - end_sources] + end_only = [source for source in ordered if source in end_sources - start_sources] + return ( + _plots.get_colour_dict(shared, "colorbrewer:Greys", value_range=(0.4, 0.45)) + | _plots.get_colour_dict( + start_only, "colorbrewer:Purples", value_range=(0.2, 0.9) + ) + | _plots.get_colour_dict(end_only, "colorbrewer:Reds", value_range=(0.2, 0.9)) + ) + + +def _profile_target( + profiles: pd.DataFrame, country: str, event_type: str, cohort: str +) -> pd.DataFrame: + """Aggregate normalized diagnostic targets for plotting.""" + if profiles.empty: + return pd.DataFrame(columns=["year", "target_mw"]) + return ( + profiles.loc[ + profiles["country_id"].eq(country) + & profiles["event_type"].eq(event_type) + & profiles["cohort"].eq(cohort) + ] + .groupby("year", as_index=False)["target_mw"] + .sum() + .sort_values("year") + ) + + +def plot_capacity_date_events( + events: pd.DataFrame, profiles: pd.DataFrame, output_path: str, category: str +) -> None: + """Plot commissioning and retirement events by date provenance.""" + title = f"Capacity-date imputation for {category.replace('_', ' ')}" + countries = sorted( + set(events["country_id"] if not events.empty else []) + | set(profiles["country_id"] if not profiles.empty else []) + ) + if not countries: + _plots.plot_empty(title, output_path) + return + + present_sources = set(events["source_type"]) + source_types = [ + source for source in _utils.DATE_SOURCE_METADATA if source in present_sources + ] + colours = _time_imputation_colours() + cols = 2 if len(countries) > 1 else 1 + rows = math.ceil(len(countries) / cols) + fig, axes = plt.subplots( + rows, cols, figsize=(cols * 7, rows * 4.5), constrained_layout=True + ) + axes_flat = np.array(axes).ravel() + + profile_kinds = ( + ("commissioning", "historical", 1, "0.15", "-"), + ("retirement", "historical", -1, "0.15", ":"), + ("commissioning", "planned", 1, "0.35", "--"), + ) + any_profile = {kind[:2]: False for kind in profile_kinds} + + for ax, country in zip(axes_flat, countries, strict=False): + country_events = events.loc[events["country_id"].eq(country)] + targets = { + (event_type, cohort): _profile_target(profiles, country, event_type, cohort) + for event_type, cohort, *_ in profile_kinds + } + year_values = set(country_events["year"].astype(int)) + for target in targets.values(): + year_values.update(target["year"].astype(int)) + years = np.array(sorted(year_values)) + if not len(years): + _plots.draw_empty(ax, country, f"No date events for {country}") + continue + + annual = ( + country_events.assign(year=country_events["year"].astype(int)) + .groupby(["year", "event_type", "source_type"])["capacity_change_mw"] + .sum() + ) + positive_bottom = np.zeros(len(years)) + negative_bottom = np.zeros(len(years)) + for source in source_types: + values = {} + for event_type in ("commissioning", "retirement"): + try: + series = annual.xs((event_type, source), level=(1, 2)) + except KeyError: + series = pd.Series(dtype=float) + values[event_type] = series.reindex(years, fill_value=0).to_numpy() + + if values["commissioning"].any(): + ax.bar( + years, + values["commissioning"], + bottom=positive_bottom, + color=colours[source], + width=0.9, + ) + if values["retirement"].any(): + ax.bar( + years, + values["retirement"], + bottom=negative_bottom, + color=colours[source], + width=0.9, + ) + positive_bottom += values["commissioning"] + negative_bottom += values["retirement"] + + for event_type, cohort, sign, colour, linestyle in profile_kinds: + target = targets[event_type, cohort] + if not target.empty: + any_profile[event_type, cohort] = True + ax.plot( + target["year"], + sign * target["target_mw"], + color=colour, + linewidth=2, + linestyle=linestyle, + ) + + ax.axhline(0, color="0.35", linewidth=0.8) + ax.set(title=country, xlabel="Year", ylabel="Annual capacity event (MW)") + ax.locator_params(axis="x", nbins=12) + ax.tick_params(axis="x", rotation=45) + ax.minorticks_off() + + for ax in axes_flat[len(countries) :]: + ax.set_visible(False) + + handles: list[Patch | Line2D] = [ + Patch( + facecolor=colours[source], + label=_utils.DATE_SOURCE_METADATA[source]["label"], + ) + for source in source_types + ] + profile_labels = { + ("commissioning", "historical"): "Commissioning profile for historic assets", + ("retirement", "historical"): "Retirement profile for historic assets", + ("commissioning", "planned"): "Commissioning profile for planned assets", + } + for event_type, cohort, _, colour, linestyle in profile_kinds: + if any_profile[event_type, cohort]: + handles.append( + Line2D( + [0], + [0], + color=colour, + linewidth=2, + linestyle=linestyle, + label=profile_labels[event_type, cohort], + ) + ) + fig.legend( + handles=handles, loc="center left", bbox_to_anchor=(1, 0.5), frameon=False + ) + fig.suptitle(title, fontsize=14) + fig.savefig(output_path, bbox_inches="tight") + plt.close(fig) + + +def _capacity_profile_diagnostics( + imputed: pd.DataFrame, profiles: pd.DataFrame, output_path: str, category: str +) -> pd.DataFrame: + """Build capacity-profile diagnostics and render their plot.""" + events = _schemas.CapacityDateEventSchema.validate( + _build_capacity_date_events(imputed) + ) + plot_capacity_date_events(events, profiles, output_path, category) + return events + + +def explore( + imputed: gpd.GeoDataFrame, output_path: str, colormap: str = "tab20" +) -> None: + """Create an HTML map for users to explore.""" + if imputed.empty: + with open(output_path, "w", encoding="utf-8") as file: + file.write("No data") + else: + imputed.explore( + column="technology", legend=True, popup=True, cmap=colormap + ).save(output_path) + + +def plot_powerplant_capacity_buildup( + plants: pd.DataFrame, output_path: str, colormap: str, category: str = "powerplant" +) -> None: + """Plot active powerplant capacity over time per country and technology.""" + title = f"Active {category} capacity by technology per country" + if plants.empty: + _plots.plot_empty(title, output_path) + return + + years = pd.Index( + range(int(plants["start_year"].min()), int(plants["end_year"].max()) + 1) + ) + countries = sorted(plants["country_id"].unique()) + technologies = sorted(plants["technology"].unique()) + colours = _plots.get_colour_dict(technologies, colormap) + cols = 2 if len(countries) > 1 else 1 + rows = math.ceil(len(countries) / cols) + fig, axes = plt.subplots( + rows, cols, figsize=(cols * 5, rows * 4), constrained_layout=True + ) + axes_flat = np.array(axes).ravel() + + for ax, country in zip(axes_flat, countries, strict=False): + country_plants = plants.loc[plants["country_id"].eq(country)] + + # Represent each lifecycle as a positive commissioning event and a + # negative retirement event. Cumulative annual changes then give active + # capacity without rescanning every plant for every plotted year. + starts = country_plants.rename(columns={"start_year": "year"}).assign( + capacity_change_mw=country_plants["output_capacity_mw"] + ) + ends = country_plants.rename(columns={"end_year": "year"}).assign( + capacity_change_mw=-country_plants["output_capacity_mw"] + ) + changes = ( + pd.concat([starts, ends]) + .groupby(["year", "technology"])["capacity_change_mw"] + .sum() + .unstack(fill_value=0) + .reindex(index=years, columns=technologies, fill_value=0) + ) + changes.cumsum().plot( + kind="bar", + stacked=True, + ax=ax, + color=[colours[technology] for technology in technologies], + legend=False, + rot=45, + ) + ax.set_title(country) + ax.set_ylabel("Capacity (MW)") + ax.locator_params(axis="x", nbins=10) + ax.minorticks_off() + + for ax in axes_flat[len(countries) :]: + ax.set_visible(False) + handles, labels = axes_flat[0].get_legend_handles_labels() + fig.legend( + handles[::-1], + labels[::-1], + loc="center left", + bbox_to_anchor=(1, 0.5), + title="Technology", + frameon=False, + ) + fig.suptitle(title, fontsize=14) + fig.savefig(output_path, bbox_inches="tight") + plt.close(fig) + + +def main() -> None: + """Main snakemake process.""" + imputed, profiles = impute_time( + plants=gpd.read_parquet(snakemake.input.relocated), + reference_capacity_df=pd.read_parquet(snakemake.input.category_capacity), + imputation=snakemake.params.imputation, + ) + schema = _schemas.build_schema(snakemake.params.tech_map, "impute") + imputed = schema.validate(imputed) + imputed.to_parquet(snakemake.output.aged) + + plot_powerplant_capacity_buildup( + imputed, + snakemake.output.histogram, + "seaborn:tab20", + snakemake.wildcards.category, + ) + explore(imputed, snakemake.output.explorer) + + match snakemake.params.imputation["method"]: + case "capacity_profile": + diagnostics = _capacity_profile_diagnostics( + imputed, + profiles, + snakemake.output.capacity_date_plot, + snakemake.wildcards.category, + ) + case method: + raise ValueError(f"Unknown time imputation method: {method}") + diagnostics.to_parquet(snakemake.output.capacity_date_events, index=False) + + +if __name__ == "__main__": + sys.stderr = open(snakemake.log[0], "w", buffering=1) + main() diff --git a/workflow/scripts/prepare_bioenergy.py b/workflow/scripts/prepare_bioenergy.py index c49343f..f3c82dd 100644 --- a/workflow/scripts/prepare_bioenergy.py +++ b/workflow/scripts/prepare_bioenergy.py @@ -63,6 +63,7 @@ def main(): }, crs=crs, ).reset_index(drop=True) + bioenergy_df = _utils.filter_noncontributing_powerplants(bioenergy_df) schema = _schemas.build_schema(technology_mapping, "prepare") schema.validate(bioenergy_df).to_parquet(snakemake.output.plants) diff --git a/workflow/scripts/prepare_fossil.py b/workflow/scripts/prepare_fossil.py index 02b2c10..f8fd7ed 100644 --- a/workflow/scripts/prepare_fossil.py +++ b/workflow/scripts/prepare_fossil.py @@ -111,7 +111,7 @@ def prepare_gem_gcpt( }, crs=crs, ).reset_index(drop=True) - coal_df = _utils.ensure_positive_capacity(coal_df) + coal_df = _utils.filter_noncontributing_powerplants(coal_df) schema = _schemas.build_schema(technology_mapping, "prepare") return schema.validate(coal_df), _schemas.FuelSchema.validate(fuels_df) @@ -156,7 +156,7 @@ def prepare_gem_gogpt( }, crs=crs, ).reset_index(drop=True) - oil_gas_df = _utils.ensure_positive_capacity(oil_gas_df) + oil_gas_df = _utils.filter_noncontributing_powerplants(oil_gas_df) schema = _schemas.build_schema(technology_mapping, "prepare") return schema.validate(oil_gas_df), _schemas.FuelSchema.validate(fuels_df) diff --git a/workflow/scripts/prepare_geothermal.py b/workflow/scripts/prepare_geothermal.py index 1e68e83..60c3bfe 100644 --- a/workflow/scripts/prepare_geothermal.py +++ b/workflow/scripts/prepare_geothermal.py @@ -41,6 +41,7 @@ def main( }, crs=crs, ).reset_index(drop=True) + geo_df = _utils.filter_noncontributing_powerplants(geo_df) schema = _schemas.build_schema(technology_mapping, "prepare") schema.validate(geo_df).to_parquet(output_plants_path) diff --git a/workflow/scripts/prepare_hydropower.py b/workflow/scripts/prepare_hydropower.py index 800aa84..99eae8c 100644 --- a/workflow/scripts/prepare_hydropower.py +++ b/workflow/scripts/prepare_hydropower.py @@ -58,6 +58,7 @@ def main(): }, crs=crs, ) + hydro_df = _utils.filter_noncontributing_powerplants(hydro_df) schema = _schemas.build_schema(technology_mapping, "prepare") schema.validate(hydro_df).to_parquet(snakemake.output.output_path) diff --git a/workflow/scripts/prepare_large_solar.py b/workflow/scripts/prepare_large_solar.py index adb98e1..014729f 100644 --- a/workflow/scripts/prepare_large_solar.py +++ b/workflow/scripts/prepare_large_solar.py @@ -155,6 +155,7 @@ def prepare_solar_utility_pv( utility_pv = pd.concat([filled_tz_df, gem_mismatch_df], ignore_index=True) # Convert to points to make further processing easier. utility_pv["geometry"] = utility_pv["geometry"].representative_point() + utility_pv = _utils.filter_noncontributing_powerplants(utility_pv) schema = _schemas.build_schema({"utility_pv": tech_name}, "prepare") return schema.validate(utility_pv) @@ -185,6 +186,7 @@ def prepare_solar_csp( "geometry": _utils.get_point_col(raw_df, "longitude", "latitude"), } ).reset_index(drop=True) + csp_df = _utils.filter_noncontributing_powerplants(csp_df) schema = _schemas.build_schema({"csp": tech_name}, "prepare") return schema.validate(csp_df) diff --git a/workflow/scripts/prepare_nuclear.py b/workflow/scripts/prepare_nuclear.py index 2d5e230..60c1849 100644 --- a/workflow/scripts/prepare_nuclear.py +++ b/workflow/scripts/prepare_nuclear.py @@ -43,6 +43,7 @@ def main( }, crs=crs, ).reset_index(drop=True) + nuclear_df = _utils.filter_noncontributing_powerplants(nuclear_df) schema = _schemas.build_schema(technology_mapping, "prepare") schema.validate(nuclear_df).to_parquet(output_plants_path) diff --git a/workflow/scripts/prepare_wind_gwpt.py b/workflow/scripts/prepare_wind_gwpt.py index b320797..f2b49d4 100644 --- a/workflow/scripts/prepare_wind_gwpt.py +++ b/workflow/scripts/prepare_wind_gwpt.py @@ -55,6 +55,7 @@ def prepare_gem_gwpt( }, crs=crs, ) + wind_df = _utils.filter_noncontributing_powerplants(wind_df) schema = _schemas.build_schema(tech_mapping, "prepare") return schema.validate(wind_df) diff --git a/workflow/scripts/prepare_wind_wemi.py b/workflow/scripts/prepare_wind_wemi.py index bde6419..3b94cc9 100644 --- a/workflow/scripts/prepare_wind_wemi.py +++ b/workflow/scripts/prepare_wind_wemi.py @@ -8,10 +8,10 @@ from typing import TYPE_CHECKING, Any import _schemas +import _utils import geopandas as gpd import numpy as np import pandas as pd -from _utils import get_point_col if TYPE_CHECKING: snakemake: Any @@ -68,10 +68,11 @@ def prepare_wemi(wemi_path: str, tech_map: dict, crs: str) -> gpd.GeoDataFrame: "start_year": start_year, "end_year": np.nan, "status": raw_df["Status"].replace(STATUS_MAPPING), - "geometry": get_point_col(raw_df, "Longitude", "Latitude", crs=crs), + "geometry": _utils.get_point_col(raw_df, "Longitude", "Latitude", crs=crs), }, crs=crs, ) + processed_df = _utils.filter_noncontributing_powerplants(processed_df) schema = _schemas.build_schema(tech_map, "prepare") return schema.validate(processed_df)