Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,7 @@ and this project adheres to [Semantic Versioning](http://semver.org/spec/v2.0.0.
- Added `FiboaDuckDBBaseConverter` for SQL-based conversion of large Parquet sources.
- Added `PerFileBaseConverter` to process multi-file sources incrementally.
- Added support for supplementary HCAT/crop mappings via `hcat_mapping_supplements`.
- Added `get_hcat_mapping()` and `hcat_lookup()` to `AddHCATMixin` to load the HCAT mapping once and look up codes or names in it.
- Added support for Esri JSON output and server-side filters in REST converters.
- Added `WFSConverterMixin` for paged downloads from WFS layers.
- Added a test guard that rejects fixture files larger than 5 MB.
Expand Down Expand Up @@ -94,6 +95,8 @@ and this project adheres to [Semantic Versioning](http://semver.org/spec/v2.0.0.
- `metrics:area` measured from the geometry is now correct in CRSs that are in metres but not equal-area, such as Web Mercator: they are reprojected to an equal-area CRS first. Source areas in hectares are converted to m² also where missing values are filled in, and empty values are filled in, not only 0. Invalid geometries are repaired before they are measured, as they are for the output, so that e.g. a self-intersecting polygon no longer gets an area of 0.
- Converters no longer publish duplicate `id`s (#282): row-numbered ids count over all source files instead of restarting per file, and ids the source repeats get a `~<n>` suffix.
- Rows missing `crop:code` are now dropped with a warning (and an error threshold), instead of failing whole conversions.
- The HCAT mapping, the German IACS crop names and `metrics:area` now use the columns returned by an overridden `get_columns()` instead of the declared `columns`.
- BR-CONAB: `metrics:area` falls back to the `Hectares` column where `area_ha` is empty.
- DE-BB:
- Excluded NBF ineligible patches with empty crop code.
- Fixed source encoding and FLIK handling.
Expand Down
17 changes: 9 additions & 8 deletions fiboa_cli/conversion/fiboa_converter.py
Original file line number Diff line number Diff line change
Expand Up @@ -48,14 +48,15 @@ def _source_column(self, target, columns=None):
return source
return None

# Also called after the migrations (post_migrate, add_hcat): overrides must be side-effect free
def get_columns(self, gdf):
columns = super().get_columns(gdf)
self._area_column = self._source_column(AREA_KEY, columns)
self._area_in_ha = self.area_is_in_ha
if self._area_column is None and self.area_calculate_missing:
self._area_added = (
self._source_column(AREA_KEY, columns) is None and self.area_calculate_missing
)
if self._area_added:
# post_migrate() measures it; the mapping keeps the column in the output
columns[AREA_KEY] = self._area_column = AREA_KEY
self._area_in_ha = False
columns[AREA_KEY] = AREA_KEY
return columns

def _variants_are_years(self):
Expand Down Expand Up @@ -159,9 +160,9 @@ def post_migrate(self, gdf):
gdf = super().post_migrate(gdf)
gdf = self._traditional_axis_order(gdf)

# get_columns() runs first and resolves both; the fallback is for direct callers
area_key = getattr(self, "_area_column", None) or self._source_column(AREA_KEY)
in_ha = getattr(self, "_area_in_ha", self.area_is_in_ha)
# the final mapping, with what an override changes after this class's get_columns()
area_key = self._source_column(AREA_KEY, self.get_columns(gdf))
in_ha = self.area_is_in_ha and not self._area_added
if area_key is None:
return gdf

Expand Down
4 changes: 1 addition & 3 deletions fiboa_cli/datasets/br_conab.py
Original file line number Diff line number Diff line change
@@ -1,7 +1,6 @@
import math
from pathlib import Path

import numpy as np
from vecorel_cli.conversion.admin import AdminConverterMixin

from ..conversion.fiboa_converter import FiboaBaseConverter
Expand Down Expand Up @@ -89,8 +88,7 @@ def file_migration(self, gdf, path, uri, layer=None):

def migrate(self, gdf):
gdf = gdf.reset_index(drop=True)
gdf["area_ha"].combine_first(gdf["Hectares"]).replace(np.nan, None, inplace=True)
gdf.loc[gdf["area_ha"] == 0, "area_ha"] = None
gdf["area_ha"] = gdf["area_ha"].combine_first(gdf["Hectares"])
gdf["cd_mun"] = gdf["cd_mun"].combine_first(gdf["CD_MUN"]).apply(fformat)
gdf["nm_mun"] = gdf["nm_mun"].combine_first(gdf["NM_MUN"]).combine_first(gdf["NM_MUNIC"])
return super().migrate(gdf)
Expand Down
7 changes: 4 additions & 3 deletions fiboa_cli/datasets/commons/de_iacs.py
Original file line number Diff line number Diff line change
Expand Up @@ -32,9 +32,10 @@ def __init__(self, *args, **kwargs):

def post_migrate(self, gdf):
gdf = super().post_migrate(gdf)
# Look up the source attribute that the converter mapped to crop:code, the same way
# AddHCATMixin.get_code_column does. Columns are still source-named at this point.
attribute = next(k for k, v in self.columns.items() if v == "crop:code")
# Columns are still source-named at this point
attribute = self._source_column("crop:code", self.get_columns(gdf))
if attribute is None:
return gdf
rows = read_data_csv(CODE_LIST_FILE)
codes = gdf[attribute]
gdf["crop:name"] = codes.map({r["original_code"]: r["original_name"] for r in rows})
Expand Down
39 changes: 26 additions & 13 deletions fiboa_cli/datasets/commons/hcat.py
Original file line number Diff line number Diff line change
Expand Up @@ -70,10 +70,24 @@ def convert(self, *args, **kwargs):
)
return super().convert(*args, **kwargs)

def get_code_column(self, gdf, code="crop:code"):
try:
attribute = next(k for k, v in self.columns.items() if v == code)
except StopIteration:
def get_hcat_mapping(self) -> list[dict]:
if self.hcat_mapping is None:
self.hcat_mapping = load_hcat_mapping(self.hcat_mapping_csv, url=self.mapping_file)
for supplement in self.hcat_mapping_supplements:
self.hcat_mapping = self.hcat_mapping + load_hcat_mapping(supplement)
return self.hcat_mapping

def hcat_lookup(self, key: str, value: str, strip: bool = False) -> dict:
"""Maps one column of the HCAT mapping to another, e.g. original_code to original_name"""
return {
(row[key].strip() if strip else row[key]): row[value] for row in self.get_hcat_mapping()
}

def get_code_column(self, gdf, code="crop:code", columns=None):
if columns is None:
columns = self.get_columns(gdf)
attribute = self._source_column(code, columns)
if attribute is None:
raise Exception(f"Misssing {code} column in converter {self.__class__.__name__}")
col = gdf[attribute]
# Should be corrected in original parser
Expand All @@ -89,25 +103,22 @@ def add_hcat(self, gdf):
# Add HCAT columns based on crop-columns
# Map to HCAT categories by using the mapping from the csv file

if self.hcat_mapping is None:
self.hcat_mapping = load_hcat_mapping(self.hcat_mapping_csv, url=self.mapping_file)
for supplement in self.hcat_mapping_supplements:
self.hcat_mapping = self.hcat_mapping + load_hcat_mapping(supplement)

columns = self.get_columns(gdf)
self.get_hcat_mapping()
from_code = "original_code"
if from_code not in self.hcat_mapping[0]:
# Some code lists have no code, only a crop_name
from_code = "original_name"
crop_code_col = self.get_code_column(gdf, "crop:name")
crop_code_col = self.get_code_column(gdf, "crop:name", columns)
else:
crop_code_col = self.get_code_column(gdf)
crop_code_col = self.get_code_column(gdf, columns=columns)

def map_to(attribute):
return {e[from_code]: e[attribute] or None for e in self.hcat_mapping}

name_col = None
if self.hcat_mapping_name_fallback and from_code == "original_code":
name_col = self.get_code_column(gdf, "crop:name")
name_col = self.get_code_column(gdf, "crop:name", columns)

def map_by_name(attribute):
# Three Walloon crops carry a trailing space in the table.
Expand All @@ -130,7 +141,9 @@ def map_by_name(attribute):

if col is not None and col.isna().any():
index = [
k for k, v in self.columns.items() if v.startswith("crop:") and k in gdf.columns
k
for k, v in columns.items()
if k in gdf.columns and any(t.startswith("crop:") for t in np.atleast_1d(v))
]
missing = gdf[col.isna()][index].drop_duplicates()
missing.reset_index(drop=True, inplace=True)
Expand Down
10 changes: 4 additions & 6 deletions fiboa_cli/datasets/de_by.py
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,7 @@
from vecorel_cli.conversion.admin import AdminConverterMixin

from ..conversion.fiboa_converter import FiboaBaseConverter
from .commons.hcat import AddHCATMixin, load_hcat_mapping
from .commons.hcat import AddHCATMixin


class Converter(AdminConverterMixin, AddHCATMixin, FiboaBaseConverter):
Expand Down Expand Up @@ -33,9 +33,7 @@ def layer_filter(self, layer: str, uri: str) -> bool:

def migrate(self, gdf: gpd.GeoDataFrame):
gdf = super().migrate(gdf)
self.hcat_mapping = load_hcat_mapping(self.hcat_mapping_csv, url=self.mapping_file)
gdf = gdf[gdf["bewirtschaftung"].isin([row["original_code"] for row in self.hcat_mapping])]

mapping_crop = {row["original_code"]: row["original_name"] for row in self.hcat_mapping}
gdf["crop:name"] = gdf["bewirtschaftung"].map(mapping_crop)
names = self.hcat_lookup("original_code", "original_name")
gdf = gdf[gdf["bewirtschaftung"].isin(names)]
gdf["crop:name"] = gdf["bewirtschaftung"].map(names)
return gdf
10 changes: 6 additions & 4 deletions fiboa_cli/datasets/de_by_block.py
Original file line number Diff line number Diff line change
Expand Up @@ -44,10 +44,14 @@ class DEBYBlockConverter(AdminConverterMixin, DEIACSMixin, WFSConverterMixin, Fi
columns = {
"geometry": "geometry",
"flik": ("flik", "id"), # derived in migrate(); unique, unlike in Baden-Württemberg
"agricultural_area_type": "crop:code", # added in file_migration(), trimmed in migrate()
"agricultural_area_type": "crop:code", # added in file_migration()
"validFrom": "determination:datetime",
}
column_migrations = {"validFrom": lambda col: pd.to_datetime(col)}
column_migrations = {
"validFrom": lambda col: pd.to_datetime(col),
# …/codelist/de.iacs/AgriculturalAreaTypeValue/AL -> AL
"agricultural_area_type": lambda col: col.str.rsplit("/", n=1).str[-1],
}

def file_migration(self, gdf, path, uri, layer=None):
# The land cover class is carried as an xlink attribute, which the GML driver does not
Expand All @@ -70,6 +74,4 @@ def migrate(self, gdf):
# The FLIK is the last dot-separated segment of the identifier URI, e.g.
# https://registry.gdi-de.org/id/de.by.inspire.invekos.lpis.aa.DEBYLI9412000570
gdf["flik"] = gdf["id"].str.rsplit(".", n=1).str[-1]
# …/codelist/de.iacs/AgriculturalAreaTypeValue/AL -> AL
gdf["agricultural_area_type"] = gdf["agricultural_area_type"].str.rsplit("/", n=1).str[-1]
return super().migrate(gdf)
16 changes: 7 additions & 9 deletions fiboa_cli/datasets/de_nds.py
Original file line number Diff line number Diff line change
Expand Up @@ -52,14 +52,12 @@ class Converter(AdminConverterMixin, AddHCATMixin, FiboaBaseConverter):
}
column_migrations = {"ANTRAGSJAH": lambda col: pd.to_datetime(col, format="%Y")}

def migrate(self, gdf):
def get_columns(self, gdf):
columns = super().get_columns(gdf)
# older editions name the crop code and the area differently
if "NC_FESTG" not in gdf.columns:
code = {"KULTURARTF", "KULTURCODE", "KC_FESTG"} & set(gdf.columns)
del self.columns["NC_FESTG"]
self.columns[code.pop()] = "crop:code"

code = next(c for c in ("KULTURARTF", "KULTURCODE", "KC_FESTG") if c in gdf.columns)
columns[code] = columns.pop("NC_FESTG")
if "AKTUELLEFL" not in gdf.columns:
del self.columns["AKTUELLEFL"]
self.columns["AKT_FL"] = "metrics:area"

return super().migrate(gdf)
columns["AKT_FL"] = columns.pop("AKTUELLEFL")
return columns
5 changes: 1 addition & 4 deletions fiboa_cli/datasets/de_sh.py
Original file line number Diff line number Diff line change
Expand Up @@ -57,10 +57,7 @@ class Converter(AdminConverterMixin, FiboaBaseConverter):
}

def migrate(self, gdf):
renames = {old: new for old, new in self.COLUMN_RENAMES.items() if old in gdf.columns}
if renames:
gdf = gdf.rename(columns=renames)
return super().migrate(gdf)
return super().migrate(gdf.rename(columns=self.COLUMN_RENAMES))

columns = {
"geometry": "geometry",
Expand Down
8 changes: 5 additions & 3 deletions fiboa_cli/datasets/de_sl_block.py
Original file line number Diff line number Diff line change
Expand Up @@ -53,9 +53,13 @@ class DESLBlockConverter(AdminConverterMixin, DEIACSMixin, WFSConverterMixin, Fi
"geometry": "geometry",
"flik": "flik", # derived in migrate(); the block reference
"id": "id", # the flik, plus a part number where a block is several polygons
"area_type": "crop:code", # added in file_migration(), trimmed in migrate()
"area_type": "crop:code", # added in file_migration()
"area": "metrics:area", # derived in migrate(); in hectares
}
column_migrations = {
# …/codelist/de.iacs/AgriculturalAreaTypeValue/GL -> GL
"area_type": lambda col: col.str.rsplit("/", n=1).str[-1],
}

def file_migration(self, gdf, path, uri, layer=None):
# The land cover class is a nested xlink attribute, which the GML driver does not guess
Expand All @@ -77,8 +81,6 @@ def migrate(self, gdf):
# "Size in ha: 0.11206, flik: DESLLI0000248744"
gdf["flik"] = gdf["description"].apply(parse_flik)
gdf["area"] = gdf["description"].apply(parse_size)
# …/codelist/de.iacs/AgriculturalAreaTypeValue/GL -> GL
gdf["area_type"] = gdf["area_type"].str.rsplit("/", n=1).str[-1]

# the source publishes several land cover records for one block, each with its own
# polygon, so the flik alone does not identify a row; the base numbers the repeats
Expand Down
6 changes: 1 addition & 5 deletions fiboa_cli/datasets/de_th.py
Original file line number Diff line number Diff line change
Expand Up @@ -62,13 +62,9 @@ class Converter(AdminConverterMixin, FiboaBaseConverter):
{"Geaendert": True, "Unveraendert": False, "Neu": None}
),
"FBI_VJ": lambda column: column.str.split(Converter.delim, regex=True),
"GEO_UPDAT": lambda column: pd.to_datetime(column, format="%d.%m.%Y", utc=True),
}

def migrate(self, gdf):
col = "GEO_UPDAT"
gdf[col] = pd.to_datetime(gdf[col], format="%d.%m.%Y", utc=True)
return super().migrate(gdf)

column_filters = {"LF": lambda col: col == "LF"}

# Schemas for the fields that are not defined in fiboa
Expand Down
10 changes: 2 additions & 8 deletions fiboa_cli/datasets/ec_fr.py
Original file line number Diff line number Diff line change
Expand Up @@ -31,11 +31,5 @@ class Converter(EuroCropsConverterMixin, FiboaBaseConverter):
column_migrations = {
"CODE_GROUP": lambda col: col.astype("string").str.strip(),
}

def migrate(self, gdf):
# SURF_PARC is rounded to 0.01 ha, so a parcel under 50 m2 reads as zero
zero = gdf["SURF_PARC"] <= 0
if zero.any():
self.info(f"Computing the area of {zero.sum():,} parcel(s) rounded down to zero")
gdf.loc[zero, "SURF_PARC"] = gdf.loc[zero, "geometry"].area / 10_000
return super().migrate(gdf)
# SURF_PARC is rounded to 0.01 ha, so a parcel under 50 m2 reads as zero and is
# measured from the geometry (area_calculate_missing)
6 changes: 2 additions & 4 deletions fiboa_cli/datasets/ee.py
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
import pandas as pd

from ..conversion.fiboa_converter import FiboaBaseConverter
from .commons.hcat import AddHCATMixin, load_hcat_mapping
from .commons.hcat import AddHCATMixin

COLUMNS = {
"geometry": "geometry",
Expand Down Expand Up @@ -48,9 +48,7 @@ class Convert(AddHCATMixin, FiboaBaseConverter):

def migrate(self, gdf):
# do a reverse mapping (from name to crop:code)
if self.hcat_mapping is None:
self.hcat_mapping = load_hcat_mapping(self.hcat_mapping_csv, url=self.mapping_file)
codes = {row["original_name"].strip(): row["original_code"] for row in self.hcat_mapping}
codes = self.hcat_lookup("original_name", "original_code", strip=True)
gdf["crop:code"] = gdf["taotletud_kultuur"].str.strip().map(codes)
return super().migrate(gdf)

Expand Down
39 changes: 13 additions & 26 deletions fiboa_cli/datasets/es.py
Original file line number Diff line number Diff line change
Expand Up @@ -31,7 +31,6 @@ class Converter(AddHCATMixin, FiboaBaseConverter):

columns = {
"geometry": "geometry",
"id": "id",
"provincia": "admin_province_code",
"municipio": "admin_municipality_code",
"dn_surface": "metrics:area",
Expand All @@ -40,6 +39,19 @@ class Converter(AddHCATMixin, FiboaBaseConverter):
"parc_supcult": "cultivation_surface",
}

# The source has no globally unique row identifier: the SIGPAC cadastral key
# plus the declaration-line index is unique per record
id_columns = (
"provincia",
"municipio",
"agregado",
"zona",
"poligono",
"parcela",
"recinto",
"ld_recinto",
)

area_is_in_ha = False

extensions = {
Expand Down Expand Up @@ -72,31 +84,6 @@ def layer_filter(self, layer: str, uri: str) -> bool:
# GPKG contains the data layer plus several codelist tables (cod_*) — only read the data.
return layer == "cultivo_declarado"

def migrate(self, gdf):
# The source has no globally unique row identifier. Build one from the SIGPAC cadastral key
# plus the declaration-line index, which is unique per record.
def part(col):
return gdf[col].astype("Int64").astype(str)

gdf["id"] = (
part("provincia").str.zfill(2)
+ "-"
+ part("municipio")
+ "-"
+ part("agregado")
+ "-"
+ part("zona")
+ "-"
+ part("poligono")
+ "-"
+ part("parcela")
+ "-"
+ part("recinto")
+ "-"
+ part("ld_recinto")
)
return super().migrate(gdf)

def get_urls(self):
if self.variant not in self.variants:
opts = ", ".join(self.variants.keys())
Expand Down
4 changes: 1 addition & 3 deletions fiboa_cli/datasets/jecam.py
Original file line number Diff line number Diff line change
Expand Up @@ -48,6 +48,7 @@ class JecamConvert(FiboaBaseConverter):
column_additions = {
"crop:code_list": CODE_LIST,
}
column_migrations = {"Irrigated": lambda col: col.astype(bool)}
extensions = {ADMIN_DIVISION, CROP_EXTENSION}
missing_schemas = {
"properties": {
Expand All @@ -71,7 +72,4 @@ def migrate(self, gdf) -> gpd.GeoDataFrame:
# but it should be reconsidered in the future
# (i.e. the removal of .fillna("").astype(str) )
gdf["crop:code"] = gdf["CropType1"].map(mapping).fillna("").astype(str)

gdf.loc[gdf["Area_ha"] == 0, "Area_ha"] = None
gdf["Irrigated"] = gdf["Irrigated"].astype(bool)
return gdf
2 changes: 1 addition & 1 deletion fiboa_cli/datasets/lv.py
Original file line number Diff line number Diff line change
Expand Up @@ -110,7 +110,7 @@ def get_urls(self):
def post_migrate(self, gdf):
gdf = super().post_migrate(gdf)
# The files carry the code without a name; the code list has it.
names = {row["original_code"].strip(): row["original_name"] for row in self.hcat_mapping}
names = self.hcat_lookup("original_code", "original_name", strip=True)
gdf["crop:name"] = self.get_code_column(gdf).str.strip().map(names)
return gdf

Expand Down
Loading
Loading