Skip to content
Merged
Show file tree
Hide file tree
Changes from 13 commits
Commits
Show all changes
23 commits
Select commit Hold shift + click to select a range
fa37d31
prototype fsmap solution
DirkEilander Aug 19, 2026
0194b93
implement fs usage for rasterio driver
DirkEilander Aug 19, 2026
0782205
Update zarr reading logic to use fsspec.FSMap for s3 private bucket a…
LuukBlom Sep 3, 2026
fbc2748
remove redundant contextmanager
LuukBlom Sep 3, 2026
db929b6
prepend pixi run to aws commands
LuukBlom Sep 3, 2026
13b3541
re-order steps so pixi is available
LuukBlom Sep 3, 2026
b66e7ce
Merge branch 'main' into fix/permission_private_bucket
LuukBlom Sep 3, 2026
defc35b
add boto3 import / HAS_BOTO3 to _compat and handle pytest skip cases
LuukBlom Sep 3, 2026
f324945
add skip marker that uses the _compat variables
LuukBlom Sep 3, 2026
6b750d3
refactor: update FSMap handling and add local filesystem check in ras…
LuukBlom Sep 3, 2026
bd0b454
add create aws profile step to sonar workflow
LuukBlom Sep 3, 2026
7abd659
add `endpoint_url` to aws config
LuukBlom Sep 3, 2026
0268d22
extract duplicated xarray reading logic from drivers into a function …
LuukBlom Sep 3, 2026
8278851
make all readers accept a filesystem as an arg to read from remote so…
LuukBlom Sep 22, 2026
05fdc90
add docs for reading from private buckets.
LuukBlom Sep 24, 2026
4bc38ea
Merge branch 'main' into fix/permission_private_bucket
LuukBlom Sep 24, 2026
57579ef
fix typo in changelog
LuukBlom Sep 24, 2026
0458702
add skipif for h5netcdf and h5py
LuukBlom Sep 24, 2026
63a147b
add the skip if to the relevant tests
LuukBlom Sep 24, 2026
8dbed36
fix refs in docs
LuukBlom Sep 24, 2026
cfe406b
implement review comments
LuukBlom Sep 28, 2026
30827b7
Merge branch 'main' into fix/permission_private_bucket
LuukBlom Sep 28, 2026
81b7e49
update verify credentials section by pointing users to the docs of aws.
LuukBlom Sep 29, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
10 changes: 10 additions & 0 deletions .github/workflows/sonar.yml
Original file line number Diff line number Diff line change
Expand Up @@ -42,6 +42,16 @@ jobs:
- uses: prefix-dev/setup-pixi@v0.10.2
with:
pixi-version: "v0.76.0"
- name: Create aws profile for minio access
env:
ACCESS_KEY: ${{ secrets.MINIO_ACCESS_KEY_READONLY_HYDROMT_DATA }}
SECRET_KEY: ${{ secrets.MINIO_SECRET_KEY_READONLY_HYDROMT_DATA }}
run: |
mkdir -p ~/.aws
pixi run aws configure set aws_access_key_id "$ACCESS_KEY" --profile hydromt-data
pixi run aws configure set aws_secret_access_key "$SECRET_KEY" --profile hydromt-data
pixi run aws configure set region eu-west-1 --profile hydromt-data
pixi run aws configure set endpoint_url https://s3.deltares.nl --profile hydromt-data
- name: Test
run: pixi run --locked test-cov
- name: SonarQube Scan
Expand Down
10 changes: 10 additions & 0 deletions .github/workflows/tests.yml
Original file line number Diff line number Diff line change
Expand Up @@ -46,6 +46,16 @@ jobs:
with:
pixi-version: "v0.76.0"
environments: ${{ matrix.dependencies }}-py${{ matrix.python-version }}
- name: Create aws profile for minio access
env:
ACCESS_KEY: ${{ secrets.MINIO_ACCESS_KEY_READONLY_HYDROMT_DATA }}
SECRET_KEY: ${{ secrets.MINIO_SECRET_KEY_READONLY_HYDROMT_DATA }}
run: |
mkdir -p ~/.aws
pixi run aws configure set aws_access_key_id "$ACCESS_KEY" --profile hydromt-data
pixi run aws configure set aws_secret_access_key "$SECRET_KEY" --profile hydromt-data
pixi run aws configure set region eu-west-1 --profile hydromt-data
pixi run aws configure set endpoint_url https://s3.deltares.nl --profile hydromt-data
- name: Test
env:
PYTEST_ADDOPTS: ${{ matrix.dependencies == 'low' && '-p no:warnings' || '' }}
Expand Down
9 changes: 9 additions & 0 deletions hydromt/_compat.py
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,7 @@
HAS_OPENPYXL = False
HAS_PYET = False
HAS_S3FS = False
HAS_BOTO3 = False

try:
import gcsfs
Expand Down Expand Up @@ -43,6 +44,14 @@
except ImportError:
pass

try:
import boto3

HAS_BOTO3 = True
except ImportError:
pass


# entrypoints in standard library only compatible from 3.10 onwards
py_version = sys.version_info
if py_version[0] >= 3 and py_version[1] >= 10:
Expand Down
6 changes: 3 additions & 3 deletions hydromt/data_catalog/drivers/dataset/dataset_driver.py
Original file line number Diff line number Diff line change
Expand Up @@ -21,7 +21,7 @@ class DatasetDriver(BaseDriver, ABC):
@abstractmethod
def read(
self, uris: list[str], *, handle_nodata: NoDataStrategy = NoDataStrategy.RAISE
) -> xr.Dataset:
) -> xr.Dataset | None:
"""
Read data from one or more URIs into an xarray Dataset.

Expand All @@ -37,8 +37,8 @@ def read(

Returns
-------
xr.Dataset
The loaded dataset.
Comment thread
LuukBlom marked this conversation as resolved.
xr.Dataset | None
The loaded dataset, or None if no data was found and the strategy allows.
"""
...

Expand Down
47 changes: 12 additions & 35 deletions hydromt/data_catalog/drivers/dataset/xarray_driver.py
Original file line number Diff line number Diff line change
Expand Up @@ -14,8 +14,9 @@
from hydromt.data_catalog.drivers.xarray_options import (
XarrayDriverOptions,
XarrayIOFormat,
_read_xarray,
)
from hydromt.error import NoDataStrategy, exec_nodata_strat
from hydromt.error import NoDataStrategy

logger = logging.getLogger(__name__)

Expand All @@ -42,7 +43,7 @@ class DatasetXarrayDriver(DatasetDriver):

def read(
self, uris: list[str], *, handle_nodata: NoDataStrategy = NoDataStrategy.RAISE
) -> xr.Dataset:
) -> xr.Dataset | None:
"""
Read zarr or netCDF data into an xarray Dataset.

Expand All @@ -60,45 +61,21 @@ def read(

Returns
-------
xr.Dataset
The dataset read from the source files.
xr.Dataset | None
The dataset read from the source files, or None if no data was found and the strategy allows.
Comment thread
LuukBlom marked this conversation as resolved.
Outdated

Raises
------
ValueError
If the provided files have mixed or unsupported extensions.
"""
preprocessor = self.options.get_preprocessor()
filtered_uris, io_format = self.options.filter_uris_by_format(uris)

# Read and merge
if io_format == XarrayIOFormat.ZARR:
datasets = [
preprocessor(xr.open_zarr(_uri, **self.options.get_kwargs()))
for _uri in filtered_uris
]
ds: xr.Dataset = xr.merge(datasets)
elif io_format == XarrayIOFormat.NETCDF4:
ds: xr.Dataset = xr.open_mfdataset(
filtered_uris,
decode_coords="all",
preprocess=preprocessor,
**self.options.get_kwargs(),
decode_timedelta=True,
)
else:
raise ValueError(
f"Unknown extension for DatasetXarrayDriver: {self.options.get_reading_ext(uris[0])}"
)

for variable in ds.data_vars:
if ds[variable].size == 0:
exec_nodata_strat(
f"No data from driver: '{self.name}' for variable: '{variable}'",
strategy=handle_nodata,
)
return None # handle_nodata == ignore
return ds
return _read_xarray(
uris=uris,
options=self.options,
filesystem=self.filesystem,
driver_name=self.name,
handle_nodata=handle_nodata,
)

def write(
self,
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -35,7 +35,7 @@ def read(
mask: Geom | None = None,
predicate: Predicate = "intersects",
metadata: SourceMetadata | None = None,
) -> xr.Dataset:
) -> xr.Dataset | None:
"""
Read in data to an xarray Dataset.

Expand Down
43 changes: 10 additions & 33 deletions hydromt/data_catalog/drivers/geodataset/xarray_driver.py
Original file line number Diff line number Diff line change
Expand Up @@ -17,8 +17,9 @@
from hydromt.data_catalog.drivers.xarray_options import (
XarrayDriverOptions,
XarrayIOFormat,
_read_xarray,
)
from hydromt.error import NoDataStrategy, exec_nodata_strat
from hydromt.error import NoDataStrategy
from hydromt.typing import (
Geom,
Predicate,
Expand Down Expand Up @@ -54,7 +55,7 @@ def read(
mask: Geom | None = None,
predicate: Predicate = "intersects",
metadata: SourceMetadata | None = None,
) -> xr.Dataset:
) -> xr.Dataset | None:
"""
Read in data to an xarray Dataset.

Expand Down Expand Up @@ -89,37 +90,13 @@ def read(
"metadata": metadata,
},
)
preprocessor = self.options.get_preprocessor()
filtered_uris, io_format = self.options.filter_uris_by_format(uris)

# Read and merge
if io_format == XarrayIOFormat.ZARR:
datasets = [
preprocessor(xr.open_zarr(_uri, **self.options.get_kwargs()))
for _uri in filtered_uris
]
ds: xr.Dataset = xr.merge(datasets)
elif io_format == XarrayIOFormat.NETCDF4:
ds: xr.Dataset = xr.open_mfdataset(
filtered_uris,
decode_coords="all",
preprocess=preprocessor,
**self.options.get_kwargs(),
decode_timedelta=True,
)
else:
raise ValueError(
f"Unknown extension for GeoDatasetXarrayDriver: {self.options.get_reading_ext(uris[0])} "
)

for variable in ds.data_vars:
if ds[variable].size == 0:
exec_nodata_strat(
f"No data from driver: '{self.name}' for variable: '{variable}'",
strategy=handle_nodata,
)
return None # handle_nodata == ignore
return ds
return _read_xarray(
uris=uris,
options=self.options,
filesystem=self.filesystem,
driver_name=self.name,
handle_nodata=handle_nodata,
)

def write(
self,
Expand Down
6 changes: 3 additions & 3 deletions hydromt/data_catalog/drivers/raster/raster_dataset_driver.py
Original file line number Diff line number Diff line change
Expand Up @@ -35,7 +35,7 @@ def read(
zoom: Zoom | None = None,
chunks: dict[str, Any] | None = None,
metadata: SourceMetadata | None = None,
) -> xr.Dataset:
) -> xr.Dataset | None:
"""
Read raster data from one or more URIs into an xarray Dataset.

Expand All @@ -61,8 +61,8 @@ def read(

Returns
-------
xr.Dataset
The loaded raster dataset.
xr.Dataset | None
The loaded raster dataset, or None if no data was found and the strategy allows.

"""
...
Expand Down
64 changes: 12 additions & 52 deletions hydromt/data_catalog/drivers/raster/raster_xarray_driver.py
Original file line number Diff line number Diff line change
Expand Up @@ -5,7 +5,6 @@
from typing import Any, ClassVar

import xarray as xr
from aiohttp.client_exceptions import ClientResponseError
from pydantic import Field

from hydromt._utils.unused_kwargs import _warn_on_unused_kwargs
Expand All @@ -18,8 +17,9 @@
from hydromt.data_catalog.drivers.xarray_options import (
XarrayDriverOptions,
XarrayIOFormat,
_read_xarray,
)
from hydromt.error import NoDataStrategy, exec_nodata_strat
from hydromt.error import NoDataStrategy
from hydromt.typing import (
Geom,
SourceMetadata,
Expand Down Expand Up @@ -59,7 +59,7 @@ def read(
zoom: Zoom | None = None,
chunks: dict[str, Any] | None = None,
metadata: SourceMetadata | None = None,
) -> xr.Dataset:
) -> xr.Dataset | None:
"""
Read zarr or netCDF raster data into an xarray Dataset.

Expand Down Expand Up @@ -87,8 +87,8 @@ def read(

Returns
-------
xr.Dataset
The merged xarray Dataset.
xr.Dataset | None
The merged xarray Dataset, or None if no data was found and the strategy allows.

Raises
------
Expand All @@ -114,37 +114,13 @@ def read(
if len(uris) == 0:
return None # handle_nodata == ignore

preprocessor = self.options.get_preprocessor()
filtered_uris, io_format = self.options.filter_uris_by_format(uris)

# Read and merge
if io_format == XarrayIOFormat.ZARR:
datasets = [
preprocessor(ds)
for ds in self._open_zarrs(filtered_uris, self.options.get_kwargs())
]
ds: xr.Dataset = xr.merge(datasets)
elif io_format == XarrayIOFormat.NETCDF4:
ds: xr.Dataset = xr.open_mfdataset(
filtered_uris,
decode_coords="all",
preprocess=preprocessor,
**self.options.get_kwargs(),
decode_timedelta=True,
)
else:
raise ValueError(
f"Unknown extension for RasterDatasetXarrayDriver: {self.options.get_reading_ext(uris[0])} "
)

for variable in ds.data_vars:
if ds[variable].size == 0:
exec_nodata_strat(
f"No data from driver: '{self.name}' for variable: '{variable}'",
strategy=handle_nodata,
)
return None # handle_nodata == ignore
return ds
return _read_xarray(
uris=uris,
options=self.options,
filesystem=self.filesystem,
driver_name=self.name,
handle_nodata=handle_nodata,
)

def write(
self,
Expand Down Expand Up @@ -194,19 +170,3 @@ def write(
data.to_netcdf(path, **write_kwargs)

return Path(path)

@staticmethod
def _open_zarrs(uris: list[str], read_kwargs: dict[str, Any]) -> list[xr.Dataset]:
"""Open multiple zarr datasets with error handling."""
datasets = []
for _uri in uris:
try:
ds = xr.open_zarr(_uri, **read_kwargs)
datasets.append(ds)
except ClientResponseError as e:
if e.status == 401:
raise PermissionError(
f"Unauthorized access to {_uri}. Check your credentials."
) from e
raise
return datasets
Loading
Loading