Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 0 additions & 1 deletion docs/getting_started.rst
Original file line number Diff line number Diff line change
Expand Up @@ -109,7 +109,6 @@ On top of that there are some optional dependecies:
- rasterstats (used in nlmod.util.zonal_statistics)
- contextily (nlmod.plot.add_background_map)
- scikit-image (used in nlmod.read.rws.calculate_sea_coverage)
- py7zr (used in nlmod.read.bofek.download_bofek_gdf)
- joblib (used in nlmod.cache)
- tqdm (used for showing progress in long-running methods)
- hydropandas (used in nlmod.read.knmi and nlmod.read.bro)
Expand Down
50 changes: 15 additions & 35 deletions nlmod/read/bofek.py
Original file line number Diff line number Diff line change
@@ -1,5 +1,5 @@
import logging
import shutil
import tempfile
import warnings
import zipfile
from io import BytesIO
Expand All @@ -17,9 +17,8 @@
def get_gdf_bofek(*args, **kwargs):
"""Get geodataframe of bofek 2020 wihtin the extent of the model.

It does so by downloading a zip file (> 100 MB) and extracting the relevant
geodatabase. Therefore the function can be slow, ~35 seconds depending on your
internet connection.
It does so by downloading a zip file and reading the included shapefile.
Therefore the function can be slow depending on your internet connection.

.. deprecated:: 0.10.0
`get_gdf_bofek` will be removed in nlmod 1.0.0, it is replaced by
Expand All @@ -44,9 +43,7 @@ def get_gdf_bofek(*args, **kwargs):

Notes
-----
An attempt was made to read the geodatabase in memory from the zip file wihtout
writing data to disk, but this was not successful. Mainly because of the difficulty
to read the geodatabase in memory.
The shapefile is extracted temporarily before reading it.
"""
warnings.warn(
"this function is deprecated and will eventually be removed, "
Expand All @@ -62,9 +59,8 @@ def get_gdf_bofek(*args, **kwargs):
def download_bofek_gdf(extent, dirname, timeout=3600):
"""Get geodataframe of bofek 2020 wihtin the extent of the model.

It does so by downloading a zip file (> 100 MB) and extracting the relevant
geodatabase. Therefore the function can be slow, ~35 seconds depending on your
internet connection.
It does so by downloading a zip file and reading the included shapefile.
Therefore the function can be slow depending on your internet connection.

Parameters
----------
Expand All @@ -84,21 +80,15 @@ def download_bofek_gdf(extent, dirname, timeout=3600):

Notes
-----
An attempt was made to read the geodatabase in memory from the zip file wihtout
writing data to disk, but this was not successful. Mainly because of the difficulty
to read the geodatabase in memory.
The shapefile is extracted temporarily before reading it.
"""
import py7zr

# set paths
dirname = Path(dirname)
fname_bofek_gdb = dirname / "GIS" / "BOFEK2020_bestanden" / "BOFEK2020.gdb"

# create directories if they do not exist
dirname.mkdir(exist_ok=True, parents=True)

# url
bofek_zip_url = "https://www.wur.nl/nl/show/bofek-2020-gis-1.htm"
bofek_zip_url = "https://bodemdata.nl/files/download/BOFEK_2020_Shape.zip"

# download zip
logger.info("Downloading BOFEK2020 GIS data (~35 seconds)")
Expand All @@ -116,24 +106,14 @@ def download_bofek_gdf(extent, dirname, timeout=3600):
progress_bar.update(len(data))
file_unzipped.write(data)

# extract geodatabase from 7z
with zipfile.ZipFile(file_unzipped, mode="r") as zf:
with py7zr.SevenZipFile(BytesIO(zf.read(zf.filelist[0])), mode="r") as z:
z.extract(
targets=["GIS/BOFEK2020_bestanden/BOFEK2020.gdb"],
path=dirname,
recursive=True,
with tempfile.TemporaryDirectory(dir=dirname) as extract_dir:
with zipfile.ZipFile(file_unzipped, mode="r") as zf:
zf.extractall(
path=extract_dir,
members=[f"BOFEK_2020.{ext}" for ext in ("shp", "shx", "dbf", "prj", "cpg")],
)

# read geodatabase
logger.debug("convert geodatabase to geojson")
gdf_bofek = gpd.read_file(fname_bofek_gdb)

# slice to extent
gdf_bofek = util.gdf_within_extent(gdf_bofek, extent)

# clean up
logger.debug("Remove geodatabase")
shutil.rmtree(fname_bofek_gdb)
gdf_bofek = gpd.read_file(Path(extract_dir) / "BOFEK_2020.shp")
gdf_bofek = util.gdf_within_extent(gdf_bofek, extent)

return gdf_bofek
1 change: 0 additions & 1 deletion pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -59,7 +59,6 @@ full = [
"rasterstats",
"contextily",
"scikit-image",
"py7zr",
"joblib",
"tqdm",
"hydropandas>=0.9.2",
Expand Down
6 changes: 4 additions & 2 deletions tests/test_005_external_data.py
Original file line number Diff line number Diff line change
Expand Up @@ -258,11 +258,13 @@ def test_get_brp():

# disable because slow (~35 seconds depending on internet connection)
@pytest.mark.skip(reason="slow")
def test_get_bofek():
def test_get_bofek(tmp_path):
# model with sea
ds = test_001_model.get_ds_from_cache("sea_model_grid_only")

# add knmi recharge to the model dataset
gdf_bofek = nlmod.read.bofek.download_bofek_gdf(ds)
gdf_bofek = nlmod.read.bofek.download_bofek_gdf(
nlmod.grid.get_extent(ds), tmp_path
)

assert not gdf_bofek.empty, "Bofek geodataframe is empty"