From b3a4a52dc66d23fd1ae983c15b814c87817f7af3 Mon Sep 17 00:00:00 2001 From: Tom Hottentot Date: Wed, 30 Sep 2026 21:53:39 +0200 Subject: [PATCH] fix bofek --- docs/getting_started.rst | 1 - nlmod/read/bofek.py | 50 ++++++++++----------------------- pyproject.toml | 1 - tests/test_005_external_data.py | 6 ++-- 4 files changed, 19 insertions(+), 39 deletions(-) diff --git a/docs/getting_started.rst b/docs/getting_started.rst index b14209ca..cfbdae52 100644 --- a/docs/getting_started.rst +++ b/docs/getting_started.rst @@ -109,7 +109,6 @@ On top of that there are some optional dependecies: - rasterstats (used in nlmod.util.zonal_statistics) - contextily (nlmod.plot.add_background_map) - scikit-image (used in nlmod.read.rws.calculate_sea_coverage) -- py7zr (used in nlmod.read.bofek.download_bofek_gdf) - joblib (used in nlmod.cache) - tqdm (used for showing progress in long-running methods) - hydropandas (used in nlmod.read.knmi and nlmod.read.bro) diff --git a/nlmod/read/bofek.py b/nlmod/read/bofek.py index 2b1a71c2..56c14811 100644 --- a/nlmod/read/bofek.py +++ b/nlmod/read/bofek.py @@ -1,5 +1,5 @@ import logging -import shutil +import tempfile import warnings import zipfile from io import BytesIO @@ -17,9 +17,8 @@ def get_gdf_bofek(*args, **kwargs): """Get geodataframe of bofek 2020 wihtin the extent of the model. - It does so by downloading a zip file (> 100 MB) and extracting the relevant - geodatabase. Therefore the function can be slow, ~35 seconds depending on your - internet connection. + It does so by downloading a zip file and reading the included shapefile. + Therefore the function can be slow depending on your internet connection. .. deprecated:: 0.10.0 `get_gdf_bofek` will be removed in nlmod 1.0.0, it is replaced by @@ -44,9 +43,7 @@ def get_gdf_bofek(*args, **kwargs): Notes ----- - An attempt was made to read the geodatabase in memory from the zip file wihtout - writing data to disk, but this was not successful. Mainly because of the difficulty - to read the geodatabase in memory. + The shapefile is extracted temporarily before reading it. """ warnings.warn( "this function is deprecated and will eventually be removed, " @@ -62,9 +59,8 @@ def get_gdf_bofek(*args, **kwargs): def download_bofek_gdf(extent, dirname, timeout=3600): """Get geodataframe of bofek 2020 wihtin the extent of the model. - It does so by downloading a zip file (> 100 MB) and extracting the relevant - geodatabase. Therefore the function can be slow, ~35 seconds depending on your - internet connection. + It does so by downloading a zip file and reading the included shapefile. + Therefore the function can be slow depending on your internet connection. Parameters ---------- @@ -84,21 +80,15 @@ def download_bofek_gdf(extent, dirname, timeout=3600): Notes ----- - An attempt was made to read the geodatabase in memory from the zip file wihtout - writing data to disk, but this was not successful. Mainly because of the difficulty - to read the geodatabase in memory. + The shapefile is extracted temporarily before reading it. """ - import py7zr - - # set paths dirname = Path(dirname) - fname_bofek_gdb = dirname / "GIS" / "BOFEK2020_bestanden" / "BOFEK2020.gdb" # create directories if they do not exist dirname.mkdir(exist_ok=True, parents=True) # url - bofek_zip_url = "https://www.wur.nl/nl/show/bofek-2020-gis-1.htm" + bofek_zip_url = "https://bodemdata.nl/files/download/BOFEK_2020_Shape.zip" # download zip logger.info("Downloading BOFEK2020 GIS data (~35 seconds)") @@ -116,24 +106,14 @@ def download_bofek_gdf(extent, dirname, timeout=3600): progress_bar.update(len(data)) file_unzipped.write(data) - # extract geodatabase from 7z - with zipfile.ZipFile(file_unzipped, mode="r") as zf: - with py7zr.SevenZipFile(BytesIO(zf.read(zf.filelist[0])), mode="r") as z: - z.extract( - targets=["GIS/BOFEK2020_bestanden/BOFEK2020.gdb"], - path=dirname, - recursive=True, + with tempfile.TemporaryDirectory(dir=dirname) as extract_dir: + with zipfile.ZipFile(file_unzipped, mode="r") as zf: + zf.extractall( + path=extract_dir, + members=[f"BOFEK_2020.{ext}" for ext in ("shp", "shx", "dbf", "prj", "cpg")], ) - # read geodatabase - logger.debug("convert geodatabase to geojson") - gdf_bofek = gpd.read_file(fname_bofek_gdb) - - # slice to extent - gdf_bofek = util.gdf_within_extent(gdf_bofek, extent) - - # clean up - logger.debug("Remove geodatabase") - shutil.rmtree(fname_bofek_gdb) + gdf_bofek = gpd.read_file(Path(extract_dir) / "BOFEK_2020.shp") + gdf_bofek = util.gdf_within_extent(gdf_bofek, extent) return gdf_bofek diff --git a/pyproject.toml b/pyproject.toml index 1c31bdbc..23f31952 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -59,7 +59,6 @@ full = [ "rasterstats", "contextily", "scikit-image", - "py7zr", "joblib", "tqdm", "hydropandas>=0.9.2", diff --git a/tests/test_005_external_data.py b/tests/test_005_external_data.py index 7dafadfd..44c8cc4f 100644 --- a/tests/test_005_external_data.py +++ b/tests/test_005_external_data.py @@ -258,11 +258,13 @@ def test_get_brp(): # disable because slow (~35 seconds depending on internet connection) @pytest.mark.skip(reason="slow") -def test_get_bofek(): +def test_get_bofek(tmp_path): # model with sea ds = test_001_model.get_ds_from_cache("sea_model_grid_only") # add knmi recharge to the model dataset - gdf_bofek = nlmod.read.bofek.download_bofek_gdf(ds) + gdf_bofek = nlmod.read.bofek.download_bofek_gdf( + nlmod.grid.get_extent(ds), tmp_path + ) assert not gdf_bofek.empty, "Bofek geodataframe is empty"