Skip to content
5 changes: 4 additions & 1 deletion docs/sphinx/source/whatsnew/v0.16.2.rst
Original file line number Diff line number Diff line change
Expand Up @@ -18,7 +18,9 @@ Bug fixes

Enhancements
~~~~~~~~~~~~

* Add ``dataset`` parameter to :py:func:`~pvlib.iotools.get_era5`
allowing for choosing between ERA5 and ERA5-Land datasets.
(:pull:`2780`)

Documentation
~~~~~~~~~~~~~
Expand All @@ -42,4 +44,5 @@ Maintenance

Contributors
~~~~~~~~~~~~
* Adam R. Jensen (:ghuser:`AdamRJensen`)

52 changes: 39 additions & 13 deletions pvlib/iotools/era5.py
Original file line number Diff line number Diff line change
Expand Up @@ -61,15 +61,17 @@ def _m_to_cm(m):


def get_era5(latitude, longitude, start, end, variables, api_key,
dataset="reanalysis-era5-single-levels-timeseries",
map_variables=True, timeout=60,
url='https://cds.climate.copernicus.eu/api/retrieve/v1/'):
"""
Retrieve ERA5 reanalysis data from the ECMWF's Copernicus Data Store.

A CDS API key is needed to access this API. Register for one at [1]_.

This API [2]_ provides a subset of the full ERA5 dataset. See [3]_ for
the available variables. Data are available on a 0.25° x 0.25° grid.
This API [2]_ provides a subset of parameters of the full ERA5 datasets,
see [3]_ for available variables. A comparison of ERA5 and ERA5-land is
available in [4]_.

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

What do you think about adding a Notes section to briefly explain the problem with coastal grid points?

Copy link
Copy Markdown
Member Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

That would be fine, but I don't have the bandwidth right. Feel free to add a suggestion


Parameters
----------
Expand All @@ -87,6 +89,10 @@ def get_era5(latitude, longitude, start, end, variables, api_key,
See [1]_ for additional options.
api_key : str
ECMWF CDS API key.
dataset : str, default ``"reanalysis-era5-single-levels-timeseries"``
The dataset to query. May be either
``"reanalysis-era5-single-levels-timeseries"`` or
``"reanalysis-era5-land-timeseries"``. See [4]_ for details.
map_variables : bool, default True
When true, renames columns of the DataFrame to pvlib variable names
where applicable. Also converts units of some variables. See variable
Expand All @@ -109,11 +115,26 @@ def get_era5(latitude, longitude, start, end, variables, api_key,
meta : dict
Metadata.

Notes
-----
Data are returned for the grid point nearest to the requested location.
The ERA5 dataset is available on a 0.25° x 0.25° grid, whereas ERA5-Land
is available on a finer 0.1° x 0.1° grid [4]_.

ERA5-Land only covers land surfaces. For locations near the coast, the
nearest grid point may lie over water, in which case ERA5-Land returns
only missing values (NaN) without raising an error. Similarly, for ERA5
the nearest grid point may be over water and therefore not
representative of conditions on land, e.g., for air temperature. Check
the returned ``meta['latitude']`` and ``meta['longitude']``, and
consider adjusting the requested coordinates slightly inland.

References
----------
.. [1] https://cds.climate.copernicus.eu/
.. [2] https://cds.climate.copernicus.eu/datasets/reanalysis-era5-single-levels-timeseries?tab=overview
.. [3] https://confluence.ecmwf.int/pages/viewpage.action?pageId=505390919
.. [4] https://confluence.ecmwf.int/display/CKB/The+family+of+ERA5+datasets
""" # noqa: E501

def _to_utc_dt_notz(dt):
Expand All @@ -140,7 +161,7 @@ def _to_utc_dt_notz(dt):
"data_format": "csv"
}
}
slug = "processes/reanalysis-era5-single-levels-timeseries/execution"
slug = f"processes/{dataset}/execution"

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

This means that if somebody copies the dataset parameter with a typo (e.g., an space) the error is harder to spot. I suggest to add a guardrail right before (if dataset not in {"...", "..."}. In addition to that, the new function signature is not backwards-compatible in those calls that provided positional parameters up to map_variables (inclusive). The guardrail would provide a meaningful message in this case, where dataset=True/False.

Copy link
Copy Markdown
Member Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Hmm, I like the idea, but overall I'm against it. My reason is that if a new compatible dataset is added, then users would not be able to use the function. For example, I was not aware that the ERA5-Land had become an option.

Indeed, it is not backward compatible, this is why I am lobbying for making most input parameters keyword only.

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I admit to not suggesting that to avoid being pedantic 😆

I don't think it's the right time for this release, but I remember PyVista has [at least in v0.46.something] a deprecation decorator that warns against the use of positional arguments (while it also allowed passing some). Just in case it's helpful in the future if iotools API was to be standardized.

response = requests.post(url + slug, json=params, headers=headers,
timeout=timeout)
submission_response = response.json()
Expand Down Expand Up @@ -182,23 +203,28 @@ def _to_utc_dt_notz(dt):
results_response = response.json()
download_url = results_response['asset']['value']['href']

# Step 4: finally, download our dataset. it's a zipfile of one CSV
# Step 4: finally, download our dataset. it's a zipfile of CSVs; some
# datasets (e.g., ERA5-Land) return one CSV per group of parameters
response = requests.get(download_url, timeout=timeout)
zipbuffer = BytesIO(response.content)
archive = zipfile.ZipFile(zipbuffer)
filename = archive.filelist[0].filename
csvbuffer = StringIO(archive.read(filename).decode('utf-8'))
df = pd.read_csv(csvbuffer)
filenames = [file.filename for file in archive.filelist]
dfs = []
for filename in filenames:
csvbuffer = StringIO(archive.read(filename).decode('utf-8'))
dfs.append(pd.read_csv(csvbuffer))

# and parse into the usual formats
metadata = submission_response['metadata'] # include messages from ECMWF
metadata['jobID'] = job_id
if not df.empty:
metadata['latitude'] = df['latitude'].values[0]
metadata['longitude'] = df['longitude'].values[0]

df.index = pd.to_datetime(df['valid_time']).dt.tz_localize('UTC')
df = df.drop(columns=['valid_time', 'latitude', 'longitude'])
if not dfs[0].empty:
metadata['latitude'] = dfs[0]['latitude'].values[0]
metadata['longitude'] = dfs[0]['longitude'].values[0]

for i, df in enumerate(dfs):
df.index = pd.to_datetime(df['valid_time']).dt.tz_localize('UTC')
dfs[i] = df.drop(columns=['valid_time', 'latitude', 'longitude'])
df = pd.concat(dfs, axis=1)

if map_variables:
# convert units and rename
Expand Down
21 changes: 19 additions & 2 deletions tests/iotools/test_era5.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,7 @@
"""

import pandas as pd
import numpy as np
import pytest
import pvlib
import requests
Expand Down Expand Up @@ -49,11 +50,27 @@ def expected():
def test_get_era5(params, expected):
df, meta = pvlib.iotools.get_era5(**params)
pd.testing.assert_frame_equal(df, expected, check_freq=False, atol=0.1)
assert meta['longitude'] == -80.0
assert meta['latitude'] == 40.0
assert np.isclose(meta['longitude'], -80.0)
assert np.isclose(meta['latitude'], 40.0)
assert isinstance(meta['jobID'], str)


@requires_ecmwf_credentials
@pytest.mark.remote_data
@pytest.mark.flaky(reruns=RERUNS, reruns_delay=RERUNS_DELAY)
def test_get_era5_land(params, expected):
params['dataset'] = "reanalysis-era5-land-timeseries"
# ghi and temp_air belong to different parameter groups and are
# returned by ERA5-Land as separate CSV files (bhi is not available)
params['variables'] = ['ghi', 'temp_air']
df, meta = pvlib.iotools.get_era5(**params)
assert np.isclose(meta['longitude'], -80.0)
assert np.isclose(meta['latitude'], 40.0)
pd.testing.assert_index_equal(df.index, expected.index)
assert set(df.columns) == {'ghi', 'temp_air'}
assert df.notna().all().all()


@requires_ecmwf_credentials
@pytest.mark.remote_data
@pytest.mark.flaky(reruns=RERUNS, reruns_delay=RERUNS_DELAY)
Expand Down
Loading