Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
268 changes: 222 additions & 46 deletions CHANGELOG.md

Large diffs are not rendered by default.

49 changes: 48 additions & 1 deletion README.md
Original file line number Diff line number Diff line change
Expand Up @@ -121,7 +121,54 @@ Check `vec describe --help` for more details.

Merges multiple Vecorel datasets to a combined Vecorel dataset:

- `vec merge ec_ee.parquet ec_lv.parquet -o merged.parquet -e https://vecorel.org/hcat-extension/v0.1.0/schema.yaml -i ec:hcat_name -i ec:hcat_code -i ec:translated_name`
- `vec merge ec_ee.parquet ec_lv.parquet -o merged.parquet`
- Only the core properties and some additional properties: `vec merge ec_ee.parquet ec_lv.parquet -o merged.parquet -i ec:hcat_name -i ec:hcat_code`
- All properties except for some: `vec merge ec_ee.parquet ec_lv.parquet -o merged.parquet -e ec:translated_name`

The merged dataset is in EPSG:4326 by default. Use `--crs` to choose another CRS,
or `--crs first` to keep the CRS of the first dataset.

Local GeoParquet files that are all in the target CRS are merged with DuckDB, so they don't need to fit into memory.
All other datasets (e.g. GeoJSON or datasets that need to be reprojected) are merged in memory.
Use `--engine` to choose the engine explicitly.

`-i` and `-e` apply to all properties, including collection-level metadata.
The geometry is required, so it can't be excluded.
Constants that differ between the datasets are moved from the collection metadata to the features.
The collection of the features is stored in a column if the datasets have multiple collections
or if a dataset has a collection column already, otherwise only in the collection metadata.

Geometries are merged as they are, except for the reprojection to the target CRS:
they are neither made valid nor converted to other geometry types (use `vec improve -g` for that),
and features with an empty or missing geometry are not dropped (see below).
The bounding boxes (`bbox`) are computed again for the merged GeoParquet file.

#### Strict and non-strict mode

By default, `vec merge` is strict: problems that make the merged dataset invalid are errors
and no dataset is written.
With `--no-strict`, `vec merge` is fail-safe: these problems are reported as warnings and
the dataset is written anyway. Check it with `vec validate` afterwards.

| Problem | Strict (default) | `--no-strict` |
| ------- | ---------------- | ------------- |
| A required property has no value for some features | Error | Warning, the property is written as nullable |
| An id repeats within a collection | Error | Warning |
| The collection of a feature can't be determined | Error | Warning, the collection is left empty |
| `-i` or `-e` removes a required property | Error | Warning |
| A required collection-only property differs between the datasets | Error | Warning, the property is removed |
| An optional collection-only property differs between the datasets | Warning, the property is removed | Warning, the property is removed |
| A feature has an empty or missing geometry | Error | Warning, the feature is kept |
| A collection-level value doesn't fit the data type of its schema | Error | Warning, the value is left empty |
| The datasets use different versions of the Vecorel specification or of an extension | Error, before any data is read | Error, before any data is read |
| The schemas of the datasets conflict | Error | Error |

Datasets with different versions of the Vecorel specification or of an extension can't be
merged, as the CLI can't upgrade them automatically yet.

The strict mode only checks what a merge can break or check with little effort.
It doesn't validate the values against the schemas, e.g. patterns or value ranges,
so a merged dataset is only valid if the values of the source datasets are valid.

Check `vec merge --help` for more details.

Expand Down
124 changes: 123 additions & 1 deletion tests/test_convert_duckdb.py
Original file line number Diff line number Diff line change
@@ -1,4 +1,5 @@
import json
import re
import sys

import geopandas as gpd
Expand All @@ -9,7 +10,7 @@
from loguru import logger

from vecorel_cli.conversion.base import BaseConverter
from vecorel_cli.conversion.duckdb import DuckDBBaseConverter
from vecorel_cli.conversion.duckdb import DuckDBBaseConverter, _constants_table
from vecorel_cli.validate import ValidateData
from vecorel_cli.vecorel.hilbert import hilbert_keys_for_table

Expand Down Expand Up @@ -621,6 +622,127 @@ def test_merge_parquet_unions_parts_that_differ(tmp_folder):
assert sorted(x for x in table.column("id").to_pylist()) == ["0", "1"]


def test_merge_parquet_hydrates_constants_the_parts_disagree_on(tmp_folder):
"""Each part keeps its constants in its collection; one the parts disagree on
goes back into the rows, as `vec merge` does, and a shared one stays put."""
parts = []
for index, region in enumerate(("north", "south")):
gdf = gpd.GeoDataFrame(
{"id": [f"{index}"], "name": ["a"], "geometry": [shapely.box(index, 0, index + 1, 1)]},
crs="EPSG:4326",
)
src = tmp_folder / f"h_src_{index}.parquet"
gdf.to_parquet(src)
Part = type(
"Part",
(DuckDBBaseConverter,),
{
**CONFIG,
"column_additions": {"region": region, "country": "NL"},
"missing_schemas": {
"properties": {
"name": {"type": "string"},
"region": {"type": "string"},
"country": {"type": "string"},
}
},
},
)
part = tmp_folder / f"h_part_{index}.parquet"
Part().convert(part, input_files={str(src): src.name})
assert "region" in json.loads(pq.read_schema(part).metadata[b"collection"])
parts.append(part)

dest = tmp_folder / "hydrated.parquet"
Converter().merge_parquet(parts, dest)

table = pq.read_table(dest)
rows = dict(zip(table.column("id").to_pylist(), table.column("region").to_pylist()))
assert rows == {"0": "north", "1": "south"}
collection = json.loads(table.schema.metadata[b"collection"])
assert "region" not in collection
assert collection["country"] == "NL" and "country" not in table.schema.names
assert ValidateData().validate(dest, num=100, schema_map={}).errors == []


def _collection_part(folder, cid, ids, area=None):
gdf = gpd.GeoDataFrame(
{
"id": ids,
"name": ["a"] * len(ids),
"geometry": [shapely.box(i, 0, i + 1, 1) for i in range(len(ids))],
},
crs="EPSG:4326",
)
src = folder / f"c_src_{cid}_{len(ids)}.parquet"
gdf.to_parquet(src)
config = {**CONFIG, "id": cid}
if area is not None:
config["column_additions"] = {"area": area}
config["missing_schemas"] = {
"properties": {"name": {"type": "string"}, "area": {"type": "double"}}
}
part = folder / f"c_part_{cid}_{len(ids)}.parquet"
type("Part", (DuckDBBaseConverter,), config)().convert(part, input_files={str(src): src.name})
return part


def test_merge_parquet_numbers_ids_per_collection(tmp_folder):
"""Ids only have to be unique per collection, so only repeats within one get numbered."""
parts = [
_collection_part(tmp_folder, "a", ["1", "2"]),
_collection_part(tmp_folder, "b", ["1", "2"]),
_collection_part(tmp_folder, "b", ["1"]),
]
dest = tmp_folder / "per_collection.parquet"
Converter().merge_parquet(parts, dest)

table = pq.read_table(dest, columns=["collection", "id"])
assert sorted(zip(table.column("collection").to_pylist(), table.column("id").to_pylist())) == [
("a", "1"),
("a", "2"),
("b", "1~1"),
("b", "1~2"),
("b", "2"),
]
assert ValidateData().validate(dest, num=100, schema_map={}).errors == []


def test_merge_parquet_hydrates_nan_as_null(tmp_folder):
"""A NaN in the collection comes from a column without values, so it's a null."""
parts = [
_collection_part(tmp_folder, "a", ["1"], area=float("nan")),
_collection_part(tmp_folder, "b", ["2"], area=1.5),
]
dest = tmp_folder / "nan.parquet"
Converter().merge_parquet(parts, dest)

table = pq.read_table(dest, columns=["id", "area"])
assert dict(zip(table.column("id").to_pylist(), table.column("area").to_pylist())) == {
"1": None,
"2": 1.5,
}


@pytest.mark.parametrize(
"value, dtype", [(300, "uint8"), ("not a date", "date-time")], ids=["overflow", "date-time"]
)
def test_constants_that_do_not_fit_their_type_are_reported(value, dtype, capsys):
log = Converter()
logger.remove()
logger.add(sys.stdout, format="{message}", level="DEBUG", colorize=False)
table = _constants_table({"x": value}, {"x": {"type": dtype}}, log=log, source="part.parquet")

# the value would fail the writer, so it's left empty
assert table.column("x").to_pylist() == [None]
message = f"x: Value {value!r} doesn't fit data type {dtype}"
out = capsys.readouterr().out
assert message in out and "(in part.parquet)" in out

with pytest.raises(ValueError, match=re.escape(message)):
_constants_table({"x": value}, {"x": {"type": dtype}}, log=log, strict=True)


def test_merge_parquet_checks_ids_that_convert_generated(tmp_folder, capsys):
"""Each part numbered its own rows from zero, so the merge has to check what
convert() is allowed to take for granted."""
Expand Down
104 changes: 104 additions & 0 deletions tests/test_encoding_geoparquet.py
Original file line number Diff line number Diff line change
Expand Up @@ -148,3 +148,107 @@ def test_read_keeps_integers_that_have_a_null(tmp_parquet_file):
from vecorel_cli.validate import ValidateData

assert ValidateData().validate(tmp_parquet_file).errors == []


def test_write_keeps_all_null_columns_in_the_data(tmp_parquet_file):
# NaN and NA aren't valid JSON, so they can't move to the collection
from geopandas import GeoDataFrame
from shapely.geometry import box

gdf = GeoDataFrame(
{
"id": ["1", "2"],
"code": pd.array([None, None], dtype="Int64"),
"area": [np.nan, np.nan],
"name": [None, None],
"geometry": [box(0, 0, 1, 1), box(1, 0, 2, 1)],
},
crs="EPSG:4326",
)
gp = GeoParquet(tmp_parquet_file)
gp.set_collection(
{
"schemas": {"c": ["https://vecorel.org/specification/v0.1.0/schema.yaml"]},
"collection": "c",
}
)
gp.write(gdf)

collection = GeoParquet(tmp_parquet_file).get_collection()
assert not {"code", "area", "name"} & set(collection)
assert {"code", "area", "name"} <= set(pq.read_schema(tmp_parquet_file).names)


def test_get_pyarrow_type_keeps_the_schema():
from vecorel_cli.parquet.types import get_pyarrow_type

schema = {"type": "object", "patternProperties": {".*": {"type": "string"}}}
assert get_pyarrow_type(schema) == pa.map_(pa.string(), pa.string())
assert get_pyarrow_type(schema) == pa.map_(pa.string(), pa.string())


def test_constant_array_parses_dates_completely():
import datetime

import pytest

from vecorel_cli.parquet.types import constant_array

schema = {"type": "date"}
for value in ["2024-01-01", "2024-01-01T00:00:00Z"]:
assert constant_array(value, schema).to_pylist() == [datetime.date(2024, 1, 1)]
for value in ["2024-01-01-invalid", "2024-01-01T12:00:00Z"]:
with pytest.raises(ValueError):
constant_array(value, schema)


def test_write_moves_a_constant_date_to_the_collection(tmp_parquet_file):
import datetime

from geopandas import GeoDataFrame
from shapely.geometry import box

gdf = GeoDataFrame(
{
"id": ["1", "2"],
"d": [datetime.date(2024, 1, 1)] * 2,
"geometry": [box(0, 0, 1, 1), box(1, 0, 2, 1)],
},
crs="EPSG:4326",
)
gp = GeoParquet(tmp_parquet_file)
gp.set_collection(
{
"schemas": {"c": ["https://vecorel.org/specification/v0.1.0/schema.yaml"]},
"collection": "c",
"schemas:custom": {"properties": {"d": {"type": "date"}}},
}
)
gp.write(gdf)

# dates are stored as ISO 8601 strings at the collection level
assert GeoParquet(tmp_parquet_file).get_collection()["d"] == "2024-01-01"


def test_read_hydrates_array_and_object_constants(tmp_parquet_file):
from geopandas import GeoDataFrame
from shapely.geometry import box

gdf = GeoDataFrame(
{"id": ["1", "2", "3"], "geometry": [box(0, 0, 1, 1)] * 3},
crs="EPSG:4326",
)
gp = GeoParquet(tmp_parquet_file)
gp.set_collection(
{
"schemas": {"c": ["https://vecorel.org/specification/v0.1.0/schema.yaml"]},
"collection": "c",
"tags": ["a", "b"],
"attrs": {"k": "v"},
}
)
gp.write(gdf, dehydrate=False)

data = GeoParquet(tmp_parquet_file).read(hydrate=True)
assert list(data["tags"]) == [["a", "b"]] * 3
assert list(data["attrs"]) == [{"k": "v"}] * 3
Loading
Loading