diff --git a/ci/envs/38-latest-conda-forge.yaml b/ci/envs/38-latest-conda-forge.yaml index c6ea712..e1d2c74 100644 --- a/ci/envs/38-latest-conda-forge.yaml +++ b/ci/envs/38-latest-conda-forge.yaml @@ -14,7 +14,7 @@ dependencies: - pytest - pytest-cov - pytest-xdist - - fsspec + # - fsspec # to have one non-minimal build without fsspec # optional - rtree - matplotlib diff --git a/geopandas/io/arrow.py b/geopandas/io/arrow.py index 256474d..2a174ad 100644 --- a/geopandas/io/arrow.py +++ b/geopandas/io/arrow.py @@ -311,13 +311,13 @@ def _to_feather(df, path, index=None, compression=None, version=None, **kwargs): feather.write_feather(table, path, compression=compression, **kwargs) -def _arrow_to_geopandas(table): +def _arrow_to_geopandas(table, metadata=None): """ Helper function with main, shared logic for read_parquet/read_feather. """ df = table.to_pandas() - metadata = table.schema.metadata + metadata = metadata or table.schema.metadata if metadata is None or b"geo" not in metadata: raise ValueError( """Missing geo metadata in Parquet/Feather file. @@ -410,6 +410,29 @@ def _get_filesystem_path(path, filesystem=None, storage_options=None): return filesystem, path +def _ensure_arrow_fs(filesystem): + """ + Simplified version of pyarrow.fs._ensure_filesystem. This is only needed + below because `pyarrow.parquet.read_metadata` does not yet accept a + filesystem keyword (https://issues.apache.org/jira/browse/ARROW-16719) + """ + from pyarrow import fs + + if isinstance(filesystem, fs.FileSystem): + return filesystem + + # handle fsspec-compatible filesystems + try: + import fsspec + except ImportError: + pass + else: + if isinstance(filesystem, fsspec.AbstractFileSystem): + return fs.PyFileSystem(fs.FSSpecHandler(filesystem)) + + return filesystem + + def _read_parquet(path, columns=None, storage_options=None, **kwargs): """ Load a Parquet object from the file path, returning a GeoDataFrame. @@ -487,7 +510,24 @@ def _read_parquet(path, columns=None, storage_options=None, **kwargs): kwargs["use_pandas_metadata"] = True table = parquet.read_table(path, columns=columns, filesystem=filesystem, **kwargs) - return _arrow_to_geopandas(table) + # read metadata separately to get the raw Parquet FileMetaData metadata + # (pyarrow doesn't properly exposes those in schema.metadata for files + # created by GDAL - https://issues.apache.org/jira/browse/ARROW-16688) + metadata = None + if table.schema.metadata is None or b"geo" not in table.schema.metadata: + try: + # read_metadata does not accept a filesystem keyword, so need to + # handle this manually (https://issues.apache.org/jira/browse/ARROW-16719) + if filesystem is not None: + pa_filesystem = _ensure_arrow_fs(filesystem) + with pa_filesystem.open_input_file(path) as source: + metadata = parquet.read_metadata(source).metadata + else: + metadata = parquet.read_metadata(path).metadata + except Exception: + pass + + return _arrow_to_geopandas(table, metadata) def _read_feather(path, columns=None, **kwargs): diff --git a/geopandas/io/tests/data/arrow/test_data_gdal350.arrow b/geopandas/io/tests/data/arrow/test_data_gdal350.arrow new file mode 100644 index 0000000..41bc5c3 Binary files /dev/null and b/geopandas/io/tests/data/arrow/test_data_gdal350.arrow differ diff --git a/geopandas/io/tests/data/arrow/test_data_gdal350.parquet b/geopandas/io/tests/data/arrow/test_data_gdal350.parquet new file mode 100644 index 0000000..6baa7e8 Binary files /dev/null and b/geopandas/io/tests/data/arrow/test_data_gdal350.parquet differ diff --git a/geopandas/io/tests/test_arrow.py b/geopandas/io/tests/test_arrow.py index 43cd077..404f7af 100644 --- a/geopandas/io/tests/test_arrow.py +++ b/geopandas/io/tests/test_arrow.py @@ -11,9 +11,9 @@ from pandas import DataFrame, read_parquet as pd_read_parquet from pandas.testing import assert_frame_equal import numpy as np import pyproj -from pyproj import CRS from shapely.geometry import box, Point, MultiPolygon + import geopandas from geopandas import GeoDataFrame, read_file, read_parquet, read_feather from geopandas.array import to_wkb @@ -650,7 +650,7 @@ def test_write_read_default_crs(tmpdir, format): read = getattr(geopandas, f"read_{format}") df = read(filename) - assert df.crs.equals(CRS("OGC:CRS84")) + assert df.crs.equals(pyproj.CRS("OGC:CRS84")) @pytest.mark.parametrize( @@ -714,3 +714,37 @@ def test_read_versioned_file(version): df = geopandas.read_parquet(DATA_PATH / "arrow" / f"test_data_v{version}.parquet") assert_geodataframe_equal(df, expected, check_crs=check_crs) + + +def test_read_gdal_files(): + """ + Verify that files written by GDAL can be read by geopandas. + Since it is currently not yet straightforward to install GDAL with + Parquet/Arrow enabled in our conda setup, we are testing with some + generated files included in the repo (using GDAL 3.5.0): + + # small dummy test dataset (not naturalearth_lowres, as this can change over time) + from shapely.geometry import box, MultiPolygon + df = geopandas.GeoDataFrame( + {"col_str": ["a", "b"], "col_int": [1, 2], "col_float": [0.1, 0.2]}, + geometry=[MultiPolygon([box(0, 0, 1, 1), box(2, 2, 3, 3)]), box(4, 4, 5,5)], + crs="EPSG:4326", + ) + df.to_file("test_data.gpkg", GEOMETRY_NAME="geometry") + and then the gpkg file is converted to Parquet/Arrow with: + $ ogr2ogr -f Parquet -lco FID= test_data_gdal350.parquet test_data.gpkg + $ ogr2ogr -f Arrow -lco FID= -lco GEOMETRY_ENCODING=WKB test_data_gdal350.arrow test_data.gpkg # noqa: E501 + """ + check_crs = Version(pyproj.__version__) >= Version("3.0.0") + + expected = geopandas.GeoDataFrame( + {"col_str": ["a", "b"], "col_int": [1, 2], "col_float": [0.1, 0.2]}, + geometry=[MultiPolygon([box(0, 0, 1, 1), box(2, 2, 3, 3)]), box(4, 4, 5, 5)], + crs="EPSG:4326", + ) + + df = geopandas.read_parquet(DATA_PATH / "arrow" / "test_data_gdal350.parquet") + assert_geodataframe_equal(df, expected, check_crs=check_crs) + + df = geopandas.read_feather(DATA_PATH / "arrow" / "test_data_gdal350.arrow") + assert_geodataframe_equal(df, expected, check_crs=check_crs)