From fbb1351f10aeee67ce351f32917f67bc08a2435e Mon Sep 17 00:00:00 2001 From: Nick Eubank Date: Thu, 28 Apr 2016 16:17:04 -0700 Subject: [PATCH 01/34] initial outline --- geopandas/geodataframe.py | 55 ++++++++++++++++++++++++++++++++++++++- tests/test_dissolve.py | 40 ++++++++++++++++++++++++++++ 2 files changed, 94 insertions(+), 1 deletion(-) create mode 100644 tests/test_dissolve.py diff --git a/geopandas/geodataframe.py b/geopandas/geodataframe.py index 78ec62b..e5fdd0d 100644 --- a/geopandas/geodataframe.py +++ b/geopandas/geodataframe.py @@ -8,7 +8,7 @@ import os import sys import numpy as np -from pandas import DataFrame, Series +from pandas import DataFrame, Series, Index from shapely.geometry import mapping, shape from shapely.geometry.base import BaseGeometry from six import string_types @@ -441,6 +441,59 @@ class GeoDataFrame(GeoPandasBase, DataFrame): plot.__doc__ = plot_dataframe.__doc__ + def dissolve(self, by=None, aggfunc=None): + """ + Dissolve geometries within `groupby` into single observation. + + Parameters + ---------- + by : string, default None + Column whose values define groups to be dissolved + aggfunc : function or string, default "first" + Aggregation function for manipulation of data associated + with each group. Passed to pandas `groupby` method. + + Returns + ------- + GeoDataFrame + """ + + if aggfunc is None: + aggfunc = lambda x: x.iloc[0,:] + + # Separate data and geometry + data = self.drop(labels=self.geometry.name, axis=1).copy() + + + groupby_plus_geometry = [self.geometry.name] + groupby_plus_geometry.append(by) + geometry = self[groupby_plus_geometry].copy() + + # Process data + aggregated_data = data.groupby(by=by).apply(aggfunc).drop(by, axis=1) + + # Process geometry + def merge_geometries(block): + + merged_geom = block.unary_union + + new_index = block.drop(self.geometry.name, axis=1).iloc[0][by] + merged_w_index = GeoSeries(merged_geom, index=Index(Series(new_index),name=by), + name=self.geometry.name) + return merged_w_index + + + aggregated_geometry = geometry.groupby(by=by, group_keys=False) + aggregated_geometry = aggregated_geometry.apply(merge_geometries) + + aggregated_geometry = GeoDataFrame(aggregated_geometry, + index=aggregated_geometry.index, + geometry=self.geometry.name) + # Recombine + aggregated = aggregated_geometry.join(aggregated_data) + aggregated = aggregated.set_geometry(self.geometry.name) + return aggregated + def _dataframe_set_geometry(self, col, drop=False, inplace=False, crs=None): if inplace: raise ValueError("Can't do inplace setting when converting from" diff --git a/tests/test_dissolve.py b/tests/test_dissolve.py new file mode 100644 index 0000000..8bac2ba --- /dev/null +++ b/tests/test_dissolve.py @@ -0,0 +1,40 @@ +from __future__ import absolute_import +import tempfile +import shutil +from shapely.geometry import Point +from geopandas import GeoDataFrame, read_file +from geopandas.tools import overlay +from .util import unittest, download_nybb + + +class TestDataFrame(unittest.TestCase): + + def setUp(self): + + nybb_filename, nybb_zip_path = download_nybb() + self.polydf = read_file(nybb_zip_path, vfs='zip://' + nybb_filename) + + self.polydf = self.polydf.rename(columns={'geometry':'myshapes'}) + self.polydf = self.polydf.set_geometry('myshapes') + + self.polydf['manhattan_bronx'] = 0 + self.polydf.loc[3:4,'manhattan_bronx']=1 + + # Merged geometry + manhattan_bronx = self.polydf.loc[3:4,] + others = self.polydf.loc[0:2,] + + + self.merged_shapes = GeoDataFrame(columns=manhattan_bronx.columns) + self.merged_shapes.loc[0, 'myshapes'] = others.geometry.unary_union + self.merged_shapes.loc[1, 'myshapes'] = manhattan_bronx.geometry.unary_union + self.merged_shapes = self.merged_shapes.set_geometry('myshapes') + + def test_geom_dissolve(self): + test = self.polydf.dissolve('manhattan_bronx') + self.assertTrue(test.geometry.name == 'myshapes') + + known = self.merged_shapes.geometry + self.assertTrue(test.geometry.geom_almost_equals(known).all()) + + From f038a868bb9de6fa5e095c6669b795cd7e8cf1a4 Mon Sep 17 00:00:00 2001 From: Joris Van den Bossche Date: Fri, 29 Apr 2016 17:40:28 +0200 Subject: [PATCH 02/34] Expose example datasets through API geopandas.datasets --- geopandas/datasets/__init__.py | 27 ++++++++++++++++++ .../naturalearth_cities.README.html | 0 .../naturalearth_cities.VERSION.txt | 0 .../naturalearth_cities.cpg | 0 .../naturalearth_cities.dbf | Bin .../naturalearth_cities.prj | 0 .../naturalearth_cities.shp | Bin .../naturalearth_cities.shx | Bin .../naturalearth_lowres.cpg | 0 .../naturalearth_lowres.dbf | Bin .../naturalearth_lowres.prj | 0 .../naturalearth_lowres.shp | Bin .../naturalearth_lowres.shx | Bin setup.py | 13 ++++++++- 14 files changed, 39 insertions(+), 1 deletion(-) create mode 100644 geopandas/datasets/__init__.py rename {doc/source/_example_data => geopandas/datasets/naturalearth_cities}/naturalearth_cities.README.html (100%) rename {doc/source/_example_data => geopandas/datasets/naturalearth_cities}/naturalearth_cities.VERSION.txt (100%) rename {doc/source/_example_data => geopandas/datasets/naturalearth_cities}/naturalearth_cities.cpg (100%) rename {doc/source/_example_data => geopandas/datasets/naturalearth_cities}/naturalearth_cities.dbf (100%) rename {doc/source/_example_data => geopandas/datasets/naturalearth_cities}/naturalearth_cities.prj (100%) rename {doc/source/_example_data => geopandas/datasets/naturalearth_cities}/naturalearth_cities.shp (100%) rename {doc/source/_example_data => geopandas/datasets/naturalearth_cities}/naturalearth_cities.shx (100%) rename {doc/source/_example_data => geopandas/datasets/naturalearth_lowres}/naturalearth_lowres.cpg (100%) rename {doc/source/_example_data => geopandas/datasets/naturalearth_lowres}/naturalearth_lowres.dbf (100%) rename {doc/source/_example_data => geopandas/datasets/naturalearth_lowres}/naturalearth_lowres.prj (100%) rename {doc/source/_example_data => geopandas/datasets/naturalearth_lowres}/naturalearth_lowres.shp (100%) rename {doc/source/_example_data => geopandas/datasets/naturalearth_lowres}/naturalearth_lowres.shx (100%) diff --git a/geopandas/datasets/__init__.py b/geopandas/datasets/__init__.py new file mode 100644 index 0000000..35b24e0 --- /dev/null +++ b/geopandas/datasets/__init__.py @@ -0,0 +1,27 @@ +import os + + +__all__ = ['available', 'get_path'] + +module_path = os.path.dirname(__file__) +available = [p for p in next(os.walk(module_path))[1] + if not p.startswith('__')] + + +def get_path(dataset): + """ + Get the path to the data file. + + Parameters + ---------- + dataset : str + The name of the dataset. See ``geopandas.datasets.available`` for + all options. + + """ + if dataset in available: + return os.path.abspath( + os.path.join(module_path, dataset, dataset + '.shp')) + else: + msg = "The dataset '{data}' is not available".format(data=dataset) + raise ValueError(msg) diff --git a/doc/source/_example_data/naturalearth_cities.README.html b/geopandas/datasets/naturalearth_cities/naturalearth_cities.README.html similarity index 100% rename from doc/source/_example_data/naturalearth_cities.README.html rename to geopandas/datasets/naturalearth_cities/naturalearth_cities.README.html diff --git a/doc/source/_example_data/naturalearth_cities.VERSION.txt b/geopandas/datasets/naturalearth_cities/naturalearth_cities.VERSION.txt similarity index 100% rename from doc/source/_example_data/naturalearth_cities.VERSION.txt rename to geopandas/datasets/naturalearth_cities/naturalearth_cities.VERSION.txt diff --git a/doc/source/_example_data/naturalearth_cities.cpg b/geopandas/datasets/naturalearth_cities/naturalearth_cities.cpg similarity index 100% rename from doc/source/_example_data/naturalearth_cities.cpg rename to geopandas/datasets/naturalearth_cities/naturalearth_cities.cpg diff --git a/doc/source/_example_data/naturalearth_cities.dbf b/geopandas/datasets/naturalearth_cities/naturalearth_cities.dbf similarity index 100% rename from doc/source/_example_data/naturalearth_cities.dbf rename to geopandas/datasets/naturalearth_cities/naturalearth_cities.dbf diff --git a/doc/source/_example_data/naturalearth_cities.prj b/geopandas/datasets/naturalearth_cities/naturalearth_cities.prj similarity index 100% rename from doc/source/_example_data/naturalearth_cities.prj rename to geopandas/datasets/naturalearth_cities/naturalearth_cities.prj diff --git a/doc/source/_example_data/naturalearth_cities.shp b/geopandas/datasets/naturalearth_cities/naturalearth_cities.shp similarity index 100% rename from doc/source/_example_data/naturalearth_cities.shp rename to geopandas/datasets/naturalearth_cities/naturalearth_cities.shp diff --git a/doc/source/_example_data/naturalearth_cities.shx b/geopandas/datasets/naturalearth_cities/naturalearth_cities.shx similarity index 100% rename from doc/source/_example_data/naturalearth_cities.shx rename to geopandas/datasets/naturalearth_cities/naturalearth_cities.shx diff --git a/doc/source/_example_data/naturalearth_lowres.cpg b/geopandas/datasets/naturalearth_lowres/naturalearth_lowres.cpg similarity index 100% rename from doc/source/_example_data/naturalearth_lowres.cpg rename to geopandas/datasets/naturalearth_lowres/naturalearth_lowres.cpg diff --git a/doc/source/_example_data/naturalearth_lowres.dbf b/geopandas/datasets/naturalearth_lowres/naturalearth_lowres.dbf similarity index 100% rename from doc/source/_example_data/naturalearth_lowres.dbf rename to geopandas/datasets/naturalearth_lowres/naturalearth_lowres.dbf diff --git a/doc/source/_example_data/naturalearth_lowres.prj b/geopandas/datasets/naturalearth_lowres/naturalearth_lowres.prj similarity index 100% rename from doc/source/_example_data/naturalearth_lowres.prj rename to geopandas/datasets/naturalearth_lowres/naturalearth_lowres.prj diff --git a/doc/source/_example_data/naturalearth_lowres.shp b/geopandas/datasets/naturalearth_lowres/naturalearth_lowres.shp similarity index 100% rename from doc/source/_example_data/naturalearth_lowres.shp rename to geopandas/datasets/naturalearth_lowres/naturalearth_lowres.shp diff --git a/doc/source/_example_data/naturalearth_lowres.shx b/geopandas/datasets/naturalearth_lowres/naturalearth_lowres.shx similarity index 100% rename from doc/source/_example_data/naturalearth_lowres.shx rename to geopandas/datasets/naturalearth_lowres/naturalearth_lowres.shx diff --git a/setup.py b/setup.py index 7dc914b..eed0c17 100644 --- a/setup.py +++ b/setup.py @@ -81,6 +81,15 @@ short_version = '%s' write_version_py() + +# get all data dirs in the datasets module +data_files = [] + +for item in os.listdir("geopandas/datasets"): + if os.path.isdir(os.path.join("geopandas/datasets/", item)) \ + and not item.startswith('__'): + data_files.append(os.path.join("datasets", item, '*')) + setup(name='geopandas', version=FULLVERSION, description='Geographic pandas extensions', @@ -89,5 +98,7 @@ setup(name='geopandas', author_email='kjordahl@enthought.com', url='http://geopandas.org', long_description=LONG_DESCRIPTION, - packages=['geopandas', 'geopandas.io', 'geopandas.tools'], + packages=['geopandas', 'geopandas.io', 'geopandas.tools', + 'geopandas.datasets'], + package_data={'geopandas': data_files}, install_requires=INSTALL_REQUIRES) From 9d95cad7cf754fff1dfda87ddf96b4e18bf65fb5 Mon Sep 17 00:00:00 2001 From: Nick Eubank Date: Fri, 29 Apr 2016 13:47:28 -0700 Subject: [PATCH 03/34] allows strings for aggfunc --- geopandas/geodataframe.py | 9 +++------ 1 file changed, 3 insertions(+), 6 deletions(-) diff --git a/geopandas/geodataframe.py b/geopandas/geodataframe.py index e5fdd0d..08c8a3d 100644 --- a/geopandas/geodataframe.py +++ b/geopandas/geodataframe.py @@ -441,7 +441,7 @@ class GeoDataFrame(GeoPandasBase, DataFrame): plot.__doc__ = plot_dataframe.__doc__ - def dissolve(self, by=None, aggfunc=None): + def dissolve(self, by=None, aggfunc='first'): """ Dissolve geometries within `groupby` into single observation. @@ -458,9 +458,6 @@ class GeoDataFrame(GeoPandasBase, DataFrame): GeoDataFrame """ - if aggfunc is None: - aggfunc = lambda x: x.iloc[0,:] - # Separate data and geometry data = self.drop(labels=self.geometry.name, axis=1).copy() @@ -470,8 +467,8 @@ class GeoDataFrame(GeoPandasBase, DataFrame): geometry = self[groupby_plus_geometry].copy() # Process data - aggregated_data = data.groupby(by=by).apply(aggfunc).drop(by, axis=1) - + aggregated_data = data.groupby(by=by).agg(aggfunc) + # Process geometry def merge_geometries(block): From 88d0b86f13bcd14dca429317b4b4ff85c3d733f2 Mon Sep 17 00:00:00 2001 From: Nick Eubank Date: Mon, 2 May 2016 15:21:43 -0700 Subject: [PATCH 04/34] More tests, cleaned up --- geopandas/geodataframe.py | 7 +++---- tests/test_dissolve.py | 34 +++++++++++++++++++++++----------- 2 files changed, 26 insertions(+), 15 deletions(-) diff --git a/geopandas/geodataframe.py b/geopandas/geodataframe.py index 08c8a3d..49e7691 100644 --- a/geopandas/geodataframe.py +++ b/geopandas/geodataframe.py @@ -461,10 +461,9 @@ class GeoDataFrame(GeoPandasBase, DataFrame): # Separate data and geometry data = self.drop(labels=self.geometry.name, axis=1).copy() - - groupby_plus_geometry = [self.geometry.name] - groupby_plus_geometry.append(by) - geometry = self[groupby_plus_geometry].copy() + groupby_plus_geometry_cols = [self.geometry.name] + groupby_plus_geometry_cols.append(by) + geometry = self[groupby_plus_geometry_cols].copy() # Process data aggregated_data = data.groupby(by=by).agg(aggfunc) diff --git a/tests/test_dissolve.py b/tests/test_dissolve.py index 8bac2ba..a0569d0 100644 --- a/tests/test_dissolve.py +++ b/tests/test_dissolve.py @@ -5,7 +5,8 @@ from shapely.geometry import Point from geopandas import GeoDataFrame, read_file from geopandas.tools import overlay from .util import unittest, download_nybb - +from pandas.util.testing import assert_frame_equal +from pandas import Index class TestDataFrame(unittest.TestCase): @@ -13,28 +14,39 @@ class TestDataFrame(unittest.TestCase): nybb_filename, nybb_zip_path = download_nybb() self.polydf = read_file(nybb_zip_path, vfs='zip://' + nybb_filename) + self.polydf = self.polydf[['geometry', 'BoroName', 'BoroCode']] self.polydf = self.polydf.rename(columns={'geometry':'myshapes'}) self.polydf = self.polydf.set_geometry('myshapes') - self.polydf['manhattan_bronx'] = 0 - self.polydf.loc[3:4,'manhattan_bronx']=1 + self.polydf['manhattan_bronx'] = 5 + self.polydf.loc[3:4,'manhattan_bronx']=6 # Merged geometry manhattan_bronx = self.polydf.loc[3:4,] others = self.polydf.loc[0:2,] - - self.merged_shapes = GeoDataFrame(columns=manhattan_bronx.columns) - self.merged_shapes.loc[0, 'myshapes'] = others.geometry.unary_union - self.merged_shapes.loc[1, 'myshapes'] = manhattan_bronx.geometry.unary_union - self.merged_shapes = self.merged_shapes.set_geometry('myshapes') + collapsed = [others.geometry.unary_union, manhattan_bronx.geometry.unary_union] + merged_shapes = GeoDataFrame({'myshapes': collapsed}, geometry='myshapes', + index=Index([5,6], name='manhattan_bronx')) + + # Different expected results + self.first = merged_shapes.copy() + self.first['BoroName'] = ['Staten Island', 'Manhattan'] + self.first['BoroCode'] = [5, 1] + + self.mean = merged_shapes.copy() + self.mean['BoroCode'] = [4,1.5] + def test_geom_dissolve(self): test = self.polydf.dissolve('manhattan_bronx') self.assertTrue(test.geometry.name == 'myshapes') + self.assertTrue(test.geom_almost_equals(self.first).all()) - known = self.merged_shapes.geometry - self.assertTrue(test.geometry.geom_almost_equals(known).all()) - + def test_first_dissolve(self): + test = self.polydf.dissolve('manhattan_bronx') + test = test.drop('myshapes', axis=1) + first = self.first.drop('myshapes', axis=1) + assert_frame_equal(first, test) From 38b96ac695fc071a0d03548e864ef05bff23c903 Mon Sep 17 00:00:00 2001 From: Nick Eubank Date: Tue, 3 May 2016 10:19:45 -0700 Subject: [PATCH 05/34] reorg, add more tests --- geopandas/geodataframe.py | 20 +++++++++----------- tests/test_dissolve.py | 12 ++++++++++++ 2 files changed, 21 insertions(+), 11 deletions(-) diff --git a/geopandas/geodataframe.py b/geopandas/geodataframe.py index 49e7691..ef6e071 100644 --- a/geopandas/geodataframe.py +++ b/geopandas/geodataframe.py @@ -451,24 +451,23 @@ class GeoDataFrame(GeoPandasBase, DataFrame): Column whose values define groups to be dissolved aggfunc : function or string, default "first" Aggregation function for manipulation of data associated - with each group. Passed to pandas `groupby` method. + with each group. Passed to pandas `groupby.agg` method. Returns ------- GeoDataFrame """ - # Separate data and geometry + # Process non-spatial component data = self.drop(labels=self.geometry.name, axis=1).copy() + aggregated_data = data.groupby(by=by).agg(aggfunc) + + # Process spatial component groupby_plus_geometry_cols = [self.geometry.name] groupby_plus_geometry_cols.append(by) geometry = self[groupby_plus_geometry_cols].copy() - # Process data - aggregated_data = data.groupby(by=by).agg(aggfunc) - - # Process geometry def merge_geometries(block): merged_geom = block.unary_union @@ -479,11 +478,10 @@ class GeoDataFrame(GeoPandasBase, DataFrame): return merged_w_index - aggregated_geometry = geometry.groupby(by=by, group_keys=False) - aggregated_geometry = aggregated_geometry.apply(merge_geometries) - - aggregated_geometry = GeoDataFrame(aggregated_geometry, - index=aggregated_geometry.index, + g = geometry.groupby(by=by, group_keys=False).apply(merge_geometries) + + aggregated_geometry = GeoDataFrame(g, + index=g.index, geometry=self.geometry.name) # Recombine aggregated = aggregated_geometry.join(aggregated_data) diff --git a/tests/test_dissolve.py b/tests/test_dissolve.py index a0569d0..58c91ac 100644 --- a/tests/test_dissolve.py +++ b/tests/test_dissolve.py @@ -1,6 +1,7 @@ from __future__ import absolute_import import tempfile import shutil +import numpy as np from shapely.geometry import Point from geopandas import GeoDataFrame, read_file from geopandas.tools import overlay @@ -50,3 +51,14 @@ class TestDataFrame(unittest.TestCase): first = self.first.drop('myshapes', axis=1) assert_frame_equal(first, test) + def test_mean_dissolve(self): + test = self.polydf.dissolve('manhattan_bronx', aggfunc='mean') + test = test.drop('myshapes', axis=1) + mean = self.mean.drop('myshapes', axis=1) + assert_frame_equal(mean, test) + + test = self.polydf.dissolve('manhattan_bronx', aggfunc=np.mean) + test = test.drop('myshapes', axis=1) + assert_frame_equal(mean, test) + + From e7bf6285a206844aa0ded3c60f45517c14de69bb Mon Sep 17 00:00:00 2001 From: Nick Eubank Date: Fri, 22 Apr 2016 13:18:00 -0700 Subject: [PATCH 06/34] move sjoin to top-level namespace --- geopandas/__init__.py | 1 + geopandas/tools/tests/test_sjoin.py | 2 +- 2 files changed, 2 insertions(+), 1 deletion(-) diff --git a/geopandas/__init__.py b/geopandas/__init__.py index 2780f17..b02d9b9 100644 --- a/geopandas/__init__.py +++ b/geopandas/__init__.py @@ -8,6 +8,7 @@ from geopandas.geodataframe import GeoDataFrame from geopandas.io.file import read_file from geopandas.io.sql import read_postgis +from geopandas.tools import sjoin # make the interactive namespace easier to use # for `from geopandas import *` demos. diff --git a/geopandas/tools/tests/test_sjoin.py b/geopandas/tools/tests/test_sjoin.py index c3b85dd..cfaaf37 100644 --- a/geopandas/tools/tests/test_sjoin.py +++ b/geopandas/tools/tests/test_sjoin.py @@ -8,7 +8,7 @@ from shapely.geometry import Point from geopandas import GeoDataFrame, read_file, base from geopandas.tests.util import unittest, download_nybb -from geopandas.tools import sjoin +from geopandas import sjoin @unittest.skipIf(not base.HAS_SINDEX, 'Rtree absent, skipping') From 54579c01bf8e07295c9a2a66e6ebbc0cc0972d71 Mon Sep 17 00:00:00 2001 From: Nick Eubank Date: Fri, 22 Apr 2016 13:53:48 -0700 Subject: [PATCH 07/34] also move up overlay --- geopandas/__init__.py | 1 + geopandas/tests/test_overlay.py | 2 +- 2 files changed, 2 insertions(+), 1 deletion(-) diff --git a/geopandas/__init__.py b/geopandas/__init__.py index b02d9b9..032ec42 100644 --- a/geopandas/__init__.py +++ b/geopandas/__init__.py @@ -9,6 +9,7 @@ from geopandas.geodataframe import GeoDataFrame from geopandas.io.file import read_file from geopandas.io.sql import read_postgis from geopandas.tools import sjoin +from geopandas.tools import overlay # make the interactive namespace easier to use # for `from geopandas import *` demos. diff --git a/geopandas/tests/test_overlay.py b/geopandas/tests/test_overlay.py index 0586fac..9aa1e3c 100644 --- a/geopandas/tests/test_overlay.py +++ b/geopandas/tests/test_overlay.py @@ -6,8 +6,8 @@ import shutil from shapely.geometry import Point from geopandas import GeoDataFrame, read_file -from geopandas.tools import overlay from geopandas.tests.util import unittest, download_nybb +from geopandas import overlay class TestDataFrame(unittest.TestCase): From 2950bc4d58425f5fe5ab39db838f2c4d5ab92c07 Mon Sep 17 00:00:00 2001 From: Nick Eubank Date: Fri, 13 May 2016 14:27:37 -0700 Subject: [PATCH 08/34] updated joining docs (#301) --- doc/source/mergingdata.rst | 62 ++++++++++++++++++++++++++++++++++++-- 1 file changed, 60 insertions(+), 2 deletions(-) diff --git a/doc/source/mergingdata.rst b/doc/source/mergingdata.rst index 66233af..9110ac1 100644 --- a/doc/source/mergingdata.rst +++ b/doc/source/mergingdata.rst @@ -1,15 +1,73 @@ +.. currentmodule:: geopandas + +.. ipython:: python + :suppress: + + import geopandas as gpd + world = gpd.GeoDataFrame().from_file('_example_data/naturalearth_lowres.shp') + cities = gpd.GeoDataFrame().from_file('_example_data/naturalearth_cities.shp') + + # For attribute join + country_shapes = world[['geometry', 'iso_a3']] + country_names = world[['name', 'iso_a3']] + + # For spatial join + countries = world[['geometry', 'name']] + countries = countries.rename(columns={'name':'country'}) + + Merging Data ========================================= +There are two ways to combine datasets in *geopandas* -- attribute joins and spatial joins. + +In an attribute join, a ``GeoSeries`` or ``GeoDataFrame`` is combined with a regular *pandas* ``Series`` or ``DataFrame`` based on a common variable. This is analogous to normal merging or joining in *pandas*. + +In a Spatial Join, observations from to ``GeoSeries`` or ``GeoDataFrames`` are combined based on their spatial relationship to one another. Attribute Joins ---------------- -[TO BE COMPLETED -- EXAMPLES OF JOINING GDF WITH PANDAS DATAFRAME] +Attribute joins are accomplished using the ``merge`` method. In general, it is recommended to use the ``merge`` method called from the spatial dataset. With that said, the stand-alone ``merge`` function will work if the GeoDataFrame is in the ``left`` argument; if a DataFrame is in the ``left`` argument and a GeoDataFrame is in the ``right`` position, the result will no longer be a GeoDataFrame. + + +For example, consider the following merge that adds full names to a ``GeoDataFrame`` that initially has only ISO codes for each country by merging it with a *pandas* ``DataFrame``. + +.. ipython:: python + + # `country_shapes` is GeoDataFrame with country shapes and iso codes + country_shapes.head() + + # `country_names` is DataFrame with country names and iso codes + country_names.head() + + # Merge with `merge` method on shared variable (iso codes): + country_shapes = country_shapes.merge(country_names, on='iso_a3') + country_shapes.head() + Spatial Joins ---------------- -[TO BE COMPLETED -- EXAMPLES OF SPATIAL JOINS] +In a Spatial Join, two geometry objects are merged based on their spatial relationship to one another. + +.. ipython:: python + + + # One GeoDataFrame of countries, one of Cities. + # Want to merge so we can get each city's country. + countries.head() + cities.head() + + # Execute spatial join + from geopandas.tools import sjoin + cities_with_country = sjoin(cities, countries, how="inner", op='intersects') + cities_with_country.head() + + +The ``op`` options determines the type of join operation to apply. ``op`` can be set to "intersects", "within" or "contains" (these are all equivalent when joining points to polygons, but differ when joining polygons to other polygons or lines). + +Note more complicated spatial relationships can be studied by combining geometric operations with spatial join. To find all polygons within a given distance of a point, for example, one can first use the ``buffer`` method to expand each point into a circle of appropriate radius, then intersect those buffered circles with the polygons in question. + From 5c2257a080bc04236f12f57829a7783edd585efb Mon Sep 17 00:00:00 2001 From: Nick Eubank Date: Tue, 17 May 2016 14:16:27 -0700 Subject: [PATCH 09/34] Docs for geometric manipulations and overlay (#308) --- doc/source/_static/overlay_operations.png | Bin 0 -> 19251 bytes doc/source/data_structures.rst | 2 +- doc/source/geometric_manipulations.rst | 74 ++++++++------------- doc/source/index.rst | 1 + doc/source/reference.rst | 18 ++--- doc/source/set_operations.rst | 76 ++++++++++++++++++++++ 6 files changed, 115 insertions(+), 56 deletions(-) create mode 100644 doc/source/_static/overlay_operations.png create mode 100644 doc/source/set_operations.rst diff --git a/doc/source/_static/overlay_operations.png b/doc/source/_static/overlay_operations.png new file mode 100644 index 0000000000000000000000000000000000000000..38beb8f60e106cfe5bb52d4d5313653325c1a1c7 GIT binary patch literal 19251 zcmeIaWmMH~*Di_((umS2Aky6kg0vtd4N{8+X^<{y5L8M+8l<}$32DRt0qGP}x_i&{ zykoy(oO8z5$)b9YAW)0*c8|(C@6RePh`|lP;QpM z_luaf;D0#>?%##~yWy%XFNIS6g=z!-1Iy`&o+}CpDf|QJ-5XY!-{DtKFx(WBWih5u zsc_Las-(q;;YZ|dvbt{4j`sEz4sIyYE*1zk3v*fz8#ilOc?D%PomcpzC@8cj3Nn(K zp1=OhzBJG}Uhdl^4KksPiuvOHqe#Qb<8`P>VCV42RA|uVTG~TmQOiE1=neLN$+OKx ze3ex=LOiacWv`Nqz|~dK*>|h%m_lqVHNw&ZzgKRdK|BR z%kA);ZE!Ts$;o;0o}IkGaei&4$rU9}B|Y_9uKe|;-%PvThjzbn)SXU@dup}T&)|h9 z%m$6z>7pKQ&h@OUtnBRViCh2k04X)!lC>gL<=;@zadUHT{k?Nvj)0O9=keyRx+2XY zbpw&h+`#s-&#y^cH-8DBa2huEeE;}#+2QZcGD>n@bCfK24Zq!_8jh?C^2zhw=x(2lVvxTlEU@jGw7DloDBqS=93)COW98sg-qgb!R3>ISowC4mMEo zp2P+{@;O@A*ie|Rw`b8Td`I#=B0@LuX%goh8k#GKc@Jmj&d;A^Z$w567VG9y3OQ7o ze#ZV;swW+WOHTICjd^u2k*}!lWQePS`{f_aFKHsuq1eRhV@2`3tx*vXvXRlzE2~3E zC}N&FfwA}HduNaKU)hWK`fdmK|NgP`c_6cd`PR+6r%7)d+fNy{1fM^D{yk42*F+Xp zG_0ztOQw;pgn*A&8TpVz^T=B*R}L@Uf0#)%!+dL|$!@AT^z862pY0ek{8DgCObkh3 zL_~6;VT*fJ!1YyQMT6Z`SGIIGO5p*y`N;8^RY_g=aMSgb|Iar;)@FUH!}yJ!Ur(B9 zYWUiGj^CU&X_x883OO2~bJ4Swm^SC+=I$OJTMZ}kz&^jrc;us=uS8zsjW6UdV*($i zl*%7TCmr_n<5SXVi=q5bHhOx-jfKu&gBJIgS2r>CE>2xhcmod7aJ}x|a8kN$WZZFe znxJBmbg_B;W#~77MeEZpn$ACt#`EuI7J4e4$9U+sdOH7Va8y0rTh;o{`s0k8oVX>f zPVO)F#^m~*?)jc?Ho!iR`uOhj!m{VN@Xkb}|(L3@a`N|O5iOe3ZM zn%{zO9_2=DXA%+<@Pv)&+V1Y|d5VdOi;+FS@l2{q+uNflFGQI&x9?>B^znNhcKega zo#Prg*{%`3iVplV)WxLz?_Ur3e%0C#dU%Q4xs#}+Fc#PBd+OqUw!xH^o=z_8WHMvd z{(fn!NJA-8Je|+#i&~YbY_r=&SNJ{u!cZ=4sxq+BXJ6bM=f<-r0h=`D|I~e2k&f%d&-&|(l)Pp$1xpyLOHy-`W z>aMhB6cbMQAm_lbG2P%eV|PIh2Nh>x)b2ab=)A03Z%2O5|AY>rng5ihX!n^ueCi0q zhs{hQ4{Tv~XQ!l@{^uolp!DYr4mX^ZdLkR2N=!{YZX)6G+U@eBF&VFa*E0Ec{^=s- zz}4E|%&nU;;+H=zN5e!(qcbJiZAM23d+(Rk*Vmt%pZ8rF#83$IC$c4xMD%ZS8Z>^& zQ;1I%4Is%dmJFWjok%aj5y$yZ(3SnMXm@uPvHmsN(q}F4i77S?4m`*!QZA!P|4Z*` z^MS5%!7SAc>+gA3;^N{7?0Vme zc^m!C1+T8IlrqFpkO5d_ChxM+pPeW@{HyN81STe?$#Ol)6OS0H4&yBrV})gN_JD-( zw9KCE`54Ss@|_hO;w?BTcKrSs98Wp7(v^ooum}rvt4v#se=7DEP>HyNkn>qqp6{?K z3VNyJ-sp8MSYtRM4w(xMq}v(tX7W8hD$pv?UKvi&z|mdtn5ZzeTI>pi2ndHXYHDT{ zTvEaYiKKgWHeFTXk|@yTM?S*QQ8iyFr9VxCPt;>8&wcZk-E1>47Z+CvWqp1%D+;?o z<2$&eOY7_Xu!4bjrnlo;13YB7Xc!oL=KTaJX~M5sT14mlj?=a8@`ysxwd@Q+FP*JT z9y{fH)*$L@O|O#_6&1oKx)?65#_y<%d_;pLmu7=7-55e96>jGx8ULp07}UCZ>SfgC zRe`5b4i~5|I)ZmfylnpbByn_fjIkjuaX$K|hfgC0d0o%R$ti+Te|c+51(M9bJGz@k zM@KTUvat_e{tY#*=pI=J?R#6(ZbJn z@aWh1<#um-^mTO7*AoU+d+*~v(q;N}l{Ga#f{0%4{~2f7n(t7}6!#N$5_^Yj-&MGR zr^it_WcU#6d4cr$sV~3ng7Wd!>_D1GjW%i(^KBY2lZkReDp7Y_At50sb_z`{t50I- z(ODp`etmg0Rbv$r7H0b5#n6ue)h6fVTNt-#R@O#7XcnryQb~VUR2p>(SA}{cRlw}u zWYxi+@u#mYVyH#k7ykSiOy)6zTRJYR&7&6^8>^JU8w!W0{?U-1#{b=T^cs$6B^4@@ zG~+b(!RB-lqjIWscGTxa??YC&ie|mhAybtipTtl)>WUOsFo)o0}tD zp*S|v^<}81q%9uXz1`ixkg<79doa=I8CQRnJ!dy=PgBt7i!JzmQ$Oa0RZ>SL!N6D> zX7G>ahzUCkyG$a)*DOiaGSi=L?n!)!Cg;PPsY~D;G5T&tHd#nSWeyw4`w+>Uc%tUBdNQ`HvAM^$q2@`MBgkN=Dn6=|1Iym|A+ zQsfk#G*Q3NiLSl99rvDi+Lv_E4~H~KJ=M<3au41AJvl$xBo}eXLycwdfJLLCqUxUh z@Ks;K?lMDJIUVE8crL@7HjPKwH?>Hl5d=^}quf9Q%25$Im*w)#Vs~RpOMKB+Y5nMs zAmT)Y9lhlFo8RTMOva0~SW8U>Vee%;J%zWnwum_N$4Y6=(?nbwz4qb9{E;8}6)d!r zegd@E_)&2C^6W6zalRdUg~M*L@)hhFpXZK=_wiO4PiY(jjhN>rKs$PETNvuUTV&F= z)Utvs#3z!Oh1+S~ABD%=o9HZP;DXa&diiHe|4vUH^+UIaHI7#1755MB@txY;mCKMt z6gX%P_rKds*T=YRd}krz89B~MCbJw)mIO$FOw^EFEk&u=85zTVk7n{&3=$P<7D*+t zYH4bJJ}5Qr$bhHjMs%ThAN?zOJl6@ivrw;w5#lM^xWgY?=I%J`MvA}-lG7J+*+#1U zI)mOB=NX|pKRnbJ8rb~K4zeMt_33&T(#5>uZsAd=^<2_WTd}a}kTVaSN>eH)Tp&yOGR@Qi;FOcaPDt}pn;OLUlC_}>g}cx=g*UOR0P7<{AB zs~B$s&tj?)aT5~n7&d(H$?o!4sa||Bb>=tQ(p`$)z2h9?PnCb(3FZ?~3fj@W_*sfA z8#eA8C+?5#^aR14@=dp=wI<~ka43T|r@6~wiTS5?pfB5i0>(Mc6 zSx|5=@*Ir)Pak;guL%@ICTnV!ecD?c+L)?gx;#Jbo0!mf)sqWJaRDxGC8B$Ms-~c~ zQn3{W7kA|~*R*=s(EU{0XEXsM-uD;2kUuvssh2))JvlswW1~_MI*}E~FCbZn#?*JsG%&ti9+QR`7aT<`x%ge{7$S7Kj zV-F^Thhsn<%=S6n!e+SfZ>EV*M@I*6i0t*%C2>U02LRMRij|)hzx0a!tuwgni1YVR z@&YGt6qo_=amNZ#{MoZbaK)Zm@L2nIGuCmvP?s{=j{E zPPsrOov3#kuvk<>gAl-^JkQ-Fg)2|Dw!1j>5I3p(HeC=m>%ZzsP!(teUi@T$=dLpA z<28@K_fxo-Z1LcVp%lv3<{};si05QGc#d&wh7T!NQAQ?2Dpuy@QV$m7$u9*Xnws!*UylAwAFo6KyliWoa4AX!l^H0oE!wwirYrn)BfwEjWVU@bK`i z6AT_6o?5%9H~`9w8!{Td(5fG^>6G{M^DX=#dXXP~XeBuR_z)sdMFGy88OU|4vjeO)Y^LzQk34rX%o4jRwf=}~8k1M;{cMXzJg&K%C=wy;2-c=`Sls1YlQrfI7Pe2byFGdC;9i-;*$k=NNcC z)#f8)|I1(QBR*^KcJ~yzueQ|Az9CeB{k_$vu);rYH3D#a3JV$Y-1HFZ?w(zDAU?|P zm+s941qDq)@J9W4&Z?G$0+l@P#YB0UxSzg-*4*sj-wC~Hb1e8N!_)Wdu)?a^X-28)?WjE`DH8!D0%ht$!)>*6S%%h6?k!%gd|^$!RX@bZv`@S?b+Y^ z@ypnYp*F$BWJlW!VGsD#V|z}Ec~6DYsPzdno*Q4YgS$L2pan1kq+|z4=o?PSU@>~i zBOk%TKjYnfeewu~DnJ#BkRDbB;)%m9N?_wO3RI|5W1@QulO*2=6{Rnh7&>jQ-k=b) z>+cS~O&ryl1W8w|!{2YP0e_rqTqc2)er@XKrCMbQAQ&TMO zj+sjV+vFnm;NeR}*n=-0pC(P#y&yGgc6&y~*ahhul7sWYryHSk0WSd8@;ffdE?3YO zO%TRSvBsYy&j%UrY(8%w1|~@4DUz11`*y9a_`8xVSMA&i1xvvVXIM!Xad{Yh*;3vU%?(1ZSv5 zC(hWIQv-umwn$n#RevNQ6A>KRP!d;-#Slre$96gQ-F|2`*5}%Mg5Dyu^y*Od*G87( zRmHO*M`L&Qm=7n%0x(*JyASD(C;|n!ZkGh??35_moKrK)Z$8g3CYxLfP-BITb{3^c zI6g!^zqfqo`(Q)ZB1k{IbAH~K&u>Vg*=Z3K4(+2^-y7`vH&#b8Ghw%RAn~D5cNbUN zt|8mSEXiOp3IEnE?q}U$c+2oWXlQ86xQoLne5Hmh?=9zB4mtaqNkUF zqeI>d7FWYu6AKFj_KdHI--059av}Fh0Kifnk1YfJI@^yL0fdrKL@evGEkejbBq*qD zp>;KZEWu+%gaEd6$QbQTb{3Hx4m{mxNwT|;f@hlH=fD>5!_b<)FilDeOwW>s?qS|(x z)z{Cj?zq24z;P~Za#9!0TIv`G?!h4-Q98I-_mH2?Y*TD z_w9!_wHE8onrc>7+9Oh{@Zg8WL$!9uDq@3@&l2}FDc8?u;|y_pRw~e}6jigzNlHp0 zlP|;t8!q3sQawIi^Zri_^Zq5=rC+%^F1k|jUR(3Z=p_dUE@va5tJKvufVcg|%QSj> ztT-B)gE#Hc9ILIr(TCDOr}I5unG!uVr1|CF-V{D7z?9aIfFQes12BR2j}`TBfMfg& zKY3Xu&&Ch!1I29~6SG`)CCS}pg2f%S`*P80WP1|j#vNbY{w2|9XeobXSDa>;)l154 z9&v<%t&}Dl3&+9|!C(%(a9f*S>nK)swQZP`l+?*$_`RLo-DhN$`Ed4=lat(jXKtYg zS!g=@r~Zl-u9UGWg-WwJZSKZF*g0Ka`JvMXRCyoT#sxUNdRr?onU{DRG<=#vF6LR; zRcQ(w*h=nO@vd**o;qPVE1h;P1u|PqkWm1 zR6MmD>>3lc4172nU10sf(&SY7@naNsPD!HTV)kLlwhV%+y&E-_BkE+@>i~PrM>D8N zNl8Ny^e$_|3ctUkFWXB!OVr|XUF<@4Ssj#xyG0a6zQ4Dpu2**#%JVW|0|D=IZib@Hi)N8k+vjCf0mWz^5pm&i3h zacgO?D~*30Jw*}i4go~n4O&s#9a(y@VC-gOU4xHu}yropTo-#@7O)qWVp9}mK(lK zPfzdS{ybq`Z4R3o26?r(G%7ilghSs18h2P7aad@x&oL(s>Bz^}FPs86%3+`WjF&Vz zEe3T}szXv=@H&v*G~m@hXOmP)ErDb!i_m=q$>HDTrn9@dHW_0vu)U?aRY4F_g{Ath z_kWMpnDya%AB_=(u6_CW({r(1r^1K@7e~P46G|!`1|V((zO8#;kb{t3qf<6~2o&GG zvbVcqPCTj)X&tC>Z{3u*JRtP<_s8x|Fn%j2IGpk513-P`n?r9PB`rOWDo_GjjR0h9 zdOkNkHsSxEhOP8fOOo^Ho<%4fy=e|K5X*<(nd6+INH_*_<=@WzJ@5q z(ZRvNUEG9Fl)Wlyq_nh0B04%GfR<4wAX$M2!Ita^nPm8(>+<5@U_*mOtXcZ!#h_GA zf~=fe9PdY!=e$BbNB2dJ!EoYlwl04Th(JoVj-oDJz%W~*_3@SQy zrO&Yg65$cgNTJ=jwYRYPY>;1@gnmutLEnXlRj{3G2A{)>@TOa}+FP1OU9+=>lF8(7 z%^(|o@U*CdtWLzD9s>88?NY{Ne@$7cE7&WUKH2^^hoiit5>ZdDzrB`)0IY$CUH2Vy zkj16??|{JE8O$0;lo=j4o)1th!yoLt+tb^7mx7|0mVFFBMWIebMi4h&XJ@CF|3$J? zDE0#ty%Sa2NEx){2GKUql4c|g=_X6S~PgG#Jp!4d=@4)R8&+EZ{GBpa6_IRN_ifo{!n_r z?xP|G{zfJ*P=D=be@YmrADiq}o4py42m$iC7= zF*Y{-#ifqFw}mVoZR>5qRvILci_Y93;RQ@D+%$}+>RRvm|30NFVJ-PMgBY}G!MuH2 zN3V_>z$yc5D|VMs8OQ_aVdq7qss9lbR46~uGjaiZpzHVXIy+$5GR{?KzB}*=MVOyQ z0t$V7w!D@8- z>iypA=YPd^dpikXE3z4PD&ToM+& zuCgY^T7tDPKwYT*T|+XtDmj^~tGj#t>{wHKM5f;!^Y#sSD+`1|S63GRKQRn^Dm5~C z#;z|kdkKM%yfj+s%dJc|j6$-9hz3OArscaW>$WH7v)@%;Fvpy)pc>W#UKqm6AJl>T z_2F(upT#4uRTCRvys@yb=#Y#*nJt$$A^o$OD?1i%mA`uY`AE{(b%oZqseJqDp_Fr8(26!64u|hn` z(&=V_K&4)+$uc6$YHIo~4x}y@0v>tFKIt-6WoE77gsw^gwvMpH?r_-6V@MY+KujcR zJw04*82t3T)Tk{Ls8#FMgosERUNlaF)6nqS_gHc7@DU!O#!XL|SrruNv44K;v}}`S zI;UT6rvSVH9J^xrLjkT#5oXLgu`*7}q@H`s35A8sp@_xe`e2Y&Sm5VqfM2CeOtRqB zwv&}~51zi)B%r5{TTUSFzuuyhC+xhG6BzF|{s?VlJ)qFm3ias#Kax))~o;kyNbD2lQ{uvrn5#JMT z;0Uq1rTD2)_m6ks4E2yjxM76NbiZ+?=Bh(d@pv&_h9Eu-oqLBZECunM=)wV@) z(PY^KAh9m#^~V!HRU!^S7p}+mK3JCo7MnN>UEu>c5+uaLDhORB4K}%Zh1b zWbS?)tjGp>pFuUl0HKRDQO*sQ10AtunzLXb!wZrfAcX-P3c+pNZqQL5x&Ii|mR+aWXDSm@)@#zUC^Sc8VF7 zuil;!2xnPI-NKpAi%{yYy9NtOQ^`R~?NZgge~?klZ3Wp1sd8XdNHFqUmmyQ@ksI%rt~lRHfX*!?68xUSV*2d}EEyW<~~VT`$wI5bhb` zeuF?&?CkIVu;7tV`kGDrwbjcN9TBQ16a-BH0kACG!)pD8Pe@i}V7uWs&nnqNoE4HP zc%+{_d)Df?D`hdBday1WkzP}GGCYza`a&S6saMAmqqOz_-VGOE9|50z?(;?(fD0?t`DG z$_!hM(&wK8eA3fnpheEU>@RNeDem|1m^PW zyAOZWk*K7YjDFKflL#-eCvI5ZTjnHOsGS; z4{)Otp>+HdD9cRIi;zUHyF)_ff*;}7fj(Zb7F0?9(&tlLBA?gJ;{uY@W}xX^0bL`- z;%eKnfB*g=X^IaQ8bJ04fTRM&Y5z-2R^;j4`fUuIPr)w+yAskTF{jkE@_;+VmPTP# zyHTCgUXmWoxb;<$Ikklo+2D|AL(VGpyNY_2D~I4-|I?iR58%PU*6exPmbAqYO@Hc@lR`bgZNpKAiMfBM4d>6)7m)B9& z=%*HLAB+TO3M9#x3mS;sdhp}qP7@Q@B&3VMLMzMTcf}OXbCKj!$>3YhDorF66PRJc zHOB>=wyT{lijy;NFgl$&FoEoQMz%+qvi~@;_YO4Za1L2!e>)#&iHuY_evF%v+;>DQ ze0qi5ErUlRmJI7?yROzwc%KdANS>E={DSLXm*&6OmL%Yef3*2{074;-02c*jxuHp0 zM>V$q*L{xE*2j?)r33w zS5$L9NO~*Tf0~BGrHKCd!F?8CVqAJ!+Q*3Gch1V$)^qx)ZPSAy`Qh01nLIY%8A1`J zNL>w-_odZ#O>K@R@SY1UkLaZmW!0zK{Y0Vl>KYwYMPP2!NKfU13Ao^A%bEB5_J*^Usr z<*XDPS8O8Ytj8h4URhax{UF>vax2Mz^0#zYijk;=qtW7SFQuAbcFk)Ms|meDF8~4H z%G-gMgopo3(wkys_HuTd;c$WC;^HPQg>bU@z=0t1^+@hLrF8LE5fPZ7u_0RqA;D~& zuwuS{Pt#cKO|FXspq-Wj@&q{hN{88p*d*+Q)vVW*kK-RK7YO`qB2r7tBxQNq;|}yM zl>1I#jru8u_{=dv!ATwDpG7SI{sT`eRq^gIs8FL6~T~AES(Ke znTjyJqJQDftaeVvXF04!rZ5aTgY9_nXUB$TrSF+Qd>YQIOM7}FQKVMgi{U*ebKg1w z0{l-FF^IKJpBJTN`E%*=VVfGWyv-vZ#-hB)mcMPNqcx|(7O;G$4PMO>&9E~~gZb<4 zlM17@%Breow0op=qMfOq(7`;x7R}(YJ@@G1@^bmGuo$>?S7&F$=kG76GBh)mDHG(S zM7oZ|b|1s-l}4-XHZgUcS@ zeb!Gs$@?Xx$(3Zixp=Pus@fihF*o#=01Ako+}$}kV$>?u-1SGs_3+eG3N(x^P6m2_ zbZD?v0eT^oC1Bb2zRBNXAvoO*SkE=_e(ss`REY8H*7jP_*BUETYOTZuORL*P-C1@e z^0f}jBS1+LDbjn=VDssVelOT$7kl)GIN~BVUm5{GHze#&HCd!R&_P7tBLW9YVd>zs zLJIGL5}oqpy{O`UhNWUl)1AQ>8{nOT=84g`-4|PC?#ahdT4g!1dCgFPZ-SngT~U(w zN@POlUjX+&!Z073ZOgQp~GBl=$)pR&5F$gkcp#=j>Xotc^0TR8aKRq6ID zspKEGg?Y>_8n*b5xdp@TOz3z~Z*qVsg-3h-BA{^Yk_UQx4TLgn_OVH;4{JUy&X~o* zN{qld-YE6kZ-vk9Z_TxpJie>@fe!$&pQ!|sJr1~)~-Y9XuGiHKFQo(_&fydU9 z*AE;V93b)OWcU2OwL_hJZSgS7&F~2rB#9zgtBBrwwIU8F5Pu9Lev`*`hLi&B+j|mg zK(#bBHGyc8ZbFy|=^++SR~0amZy5}(koOl&CRPQmz zpO|xhBD-y^vFzjP>kGAjO%!A2jAF3PEBe@q%}MQ=n2w!cnF zB1Ll9%-f5rt|#R*xEs|Gz~`_Lcv-2WWO<;7)k(;}_JF@rjd_>h;ltE=`)TzC2gAW6 zuFQo1RnVL*Mp6^eYjk%02JKOuD|odQ0n@+u_wNw6R|Mp=mJ+3Hck~eKrG%}tGTnA_ zL;D4lg*w=+tkDJVBv^uP_L`*mqp}vFO z^v={v0b3Z2VfghC}Yb&@y3E?Ork30b7 zB*(`SKCic1JUQ5V_~68tO2R)4>{C^48|va0JJRU%SjVZuXyR;iO*f;R1iqNEIa+y( z78@&XaWfuuUG4pNR2=4L^%LMz}Avl)hUJpUZ1L7?}Je`C(VBXPFbJDEG(S$6hs- zk?qnwHb1G$lm||kID&z!@8GK|R>;R+Zwm?v%Dc}&{Ucn zDZ0!a_sw^|ncb&5N_=UF;3k7W$^vQ@iD=>N`M_nQoX-$>{&ougmc!ZQ!&2pi0cwjq zN28GLgZg?m>zzl&zPq7dB0B!e&pdG?4h{fy#3csetw~-9HQ_DCPd|QWym}JkI*9ft zsk4R%qe!QMW^r*5d$nWE*tm6}H--|hh2X%Pv6ePE_Z!6F;o)jmr*&EduBl9DsonWp zGjF5kEVQl>BNsdDNjkB`w9iqM>aP(wn zr_wHELgR>xvM-qus_743#{F|%H#EHJgI%7gUK19fxou|K_ExwNwTr_OwQhag8ujK) zo3B|nK&vzdehzr<2j4732v3>dZvIwZX6c+4;HHsGn8-2ulrwiU)-Qw$ydBaW_Liqq z>!H|3=}|_vf0MB|q@23jVcSm|cX50*Zd%s6=g7a_hRYcz6E(Io&WY(NtO!xw6kzNV zeElWx#?5tT9O8`S0FYj3gHz(vc71|9ndC`6+kL(7A{ahNWK2?(LPsl7whm}yg3gkqg#>tsmG__&83r{qyB4Tdpi z(@^k%YoYE(ex$kp?+E?5zU-|E(}h1+$-gLWB~>x$44$zYSEr7OpDVFNFwDN-Yud$p zk@w>c)UhU4`DOd7os(TN=wfBTQf&jq-W-+uylPfQ4ycH@6#Nl^8%)CDZB+OFc9oy5 zUAMx1g7+W~QV(FfkgP0PcpKmdeoB$X$Fp+aPUCq!l{)sVtd*E)u}jxWo3u{zwAD0R zpLAk`B={YiOkC6QJ5$!lPAqBi0PE^c zMLGq}mk*bxo+65le}L4dk|Abum@?D+^7W0Ix6~k^f%8&niKnP^ka8e_g%Ekpz5Qh~ zx`s(#?tMB+SvwX(Hc@c&J%mL9|id&guy!iS;vXH&a z!F7S`EjSGyLS1xv`SJm{-Si7h-FbfteBt-r$NiUv!zUh&N}Q!4)NlJLUR$Ca{XD7j zr}IXUQ5mD(#km@I1V9t-QLPSy$( z2I`53?m>iRsqQn;#Gbs51DG$HgI7;M9 zwzYxLF?HaPT%wlOIqAz(e$5J_^kS_Nq`NT;kFqzi?zfG|5jx73spyH+r>wW{i8Ju> z-VKC+>V11}w8`~18L%h-HuID}XyZH^Hu|EPb!pys_@Sy6zXJW^Z1ySUlf+xx`}WVa86}IQvIQB{df$VVt z-xio!aE-;)**)WRX6C&|-UqZWD#CbqA`~609G|5> z#W+kp@fuWku+^qG&P#Q-tJT@FpT93%c{137i2M%wExVjaOLvsxVWC@%li{! zq~{Sxw&ixCDpPcD&|qZ2#WII(Z>t_;9FS%2>J)BnZq^0%EQz#yija2t$27@HEvU+q zL1A#|>U8?`;NEpjc<&EC9sHYIZ|l><1rs0pxE*~p7mJy zl77xL-JZ|L&z}JEKr!HedDiahl`3d&>wfyT<28eREQfxr^~?kS=d+>BOur-YQ(ie_ z&P5JH0dG(K(9L>g6Zm|flEtvqSMWf>#~xYyakew0tt;3s{K6^ss`q{+O3@Fd5u2{t z<&t*MyZgC+7Y=^85jhnNv<|ev5K|mUM*qEv<%8bb(vLG{!WYZm8n-ij-4m|1dPi%_ z2c%Bb1tR=yE}GS{r2ghb5_q@jk>));$K^(6 z7@_*}v#cbPLHnaM$8ujv`Zv3@o%`q#SFz9T%m0AUtoUgJ(j^bR7g?a2Zr;Mp0_oB6 zLkl>?qv5;7jg2B&nI(lm{@$2bmyF66oJucl(j7-Pta9A&68CF+HsjX+cfLl!vQ_ho zXoo^?;NtO}8y;NU_1nht?j|XZzQy+D=F`y73{y>O5HdwxOoydD(zHFD*dBlLH}_bi z)$rhq;>Pb$0C%iann-^;(}NQS*gqn~rj zV`!K&7oB$R+E@L@$sISepJ%sC6w6r-{!Q_19l_w&!pcel!&@Y^A|oevmxksIOhdrw zabI7*(qEO4VKap+}*>TNXP z>RX)G{lf>(**^7m)c7+Qm;GV*_O#=Dy#ziR2QMcN0W|za zQdM0R3j60@@<#u@NBmwIUGf(%60DRR{uXYLJ)w>%?hzhCXngzS^w^io*BEv;mQ|)C zR0lQ7^aaFE7J`q@T9{PQ7A{WrAtFD3G2i862Z6kW(6R39?9hUs2(w>bOxNt5fm5PN zUD^0vARRBRksO&sfG0I-L2LWwj02_*F12Bn%*|z*Re@aEf8VwpC%we!0vwj}EWwn8 zm~Xyv?)boCCKqExsK8L~{%%jzuf~PV)A<1t4hcI7lGojC%Vwt!U{%L|SAce)dvWG- zC;P3J)8;W3t}EH5DAQKGv&McL`!7$d&p`D4$dScC8ooh!xOV9Pm<6o~RUo8v=dBqL zl$lx3kALHb;*jl#SjdbIJAkP)hb3t^mhy-6RHEC7SifyQl8?Xr#Hpk zOZ2|LI`W`|LeE=Up(gji;w<)yyXAv?f(Cb`M<`}fA0s`RND|&BD>yQ@Y#d@kdW2$ zy0VfR>DmJE624Yuhit(f2A)4f)g`SrR3b!Ds9smNocFZw_Z+tV%=_htwqgg3vS4^g zGz4heid_7qaAFhf@~+|dUpDWPch#vZ8jRqC%R7)6ehf1sPN{n=8e&_lxL#JDqb$@g zBh5iG0|DTUCjZK#RTP);p^vn8zru_n@K-cEzfFn4TI<5<=zx{LUNW8p^Hs1Z!GFyS zYvf?P|IZU5CvE?4-)L$tO9Qpu4|8IIf`Z`dTUc5OYG`P9|8pKlRDS#EdMyG|froDE zfv;abow*JN3Ie43qQ>d_^z@J-Z!bACR8`RdxP$jy7hsIs0*9xkw>Q!Qxwkg*6hL7z zpA|j~eIS)xZ)0y9V0v<(WWpz@Rho1LTuwoTyI#GP0lyOqn7YxiiI7kc&}RlH7~px& zZScnjmvIjKx@NgS3W%eGWMn55^P~#6`oBM$k}GL~-hYD|M*4oh1oHWpV?shgGl1l2 zooScTuKk%39i1a1BebT4Uqi%_RD(;oAVjc=H?7A{EC#teT;gFkv1k!llN4wyJ#f* zl|aTvdN{xty?xt#;_64VJT^|K3 z0gnU&CCSib>;dz;xUzDPIhqaRCIH(|UXhH;1G=ZC$OGZs+9xg{4z~XwN6W8Wf2FszJ(~C3#~aO* zU}9w*T$&RdSrv$f=m31$2gM%=L_az|A_)3&`Rac&)zR1_Gpjj*ju+!-re-z-oVQ_w$Ko zEBSfUt^t)C1x$$NauV0D3ys$tj(dK0%J%8CHMPN?Nh##iYSI~mv&3Pb%zMm4B=B3reQKu~_Z zEQ6q}k}^+DM?t}kK>l3-P-dXRbarzyhoQef=L#Bcqa z9UoG}ACFfUOMr7kB;ZO69=H$6JM09n_klGGO(er)22hvCDMx4%_5(K4{gjCM?)7FK zGWTWIo4mkBi>0+LWMoCnwvA!7t<5&!FH5!dE8=fom9EE?fjA<>C?Z>;gR9}bSuka z2kmD>L?k3RRaJM~Tc}|n!8y#6jn4O}@G;u2uY6+4!8!B>gf3ln65rj=q}Dbzoxomm z!)Sc+h$XvjB?_>4siP~SNxr9VB#aM8`Q}>jQGk01o0&0yRiP^=C{*f~om}jXI)mu$ zznKeBg@OX}ugiQ}U?GjDt>s5io3;Q+!sy4jZN?(k+l**|_Os6QR#~#vh&9>$;5V-> zg7w$!au%F~F$0mddXNG(+#Y&?LW8;hv5dDW83(fuR^S-kh8fm4--Go#-QC@Ky1Iq! z1cwE#oEM zmIG(oWb=a&tVes0-OxUP8)FBI?qGgk8yXtI3e5CBv-vdld&uMBG^ZhX_xDRAM|Y#L zu1?GE+Go41uct?*yL~R+c?mjcw8yf~E_&N4st>=Pn`R&w-ahW72)H!Syw|mGDEF<^ z+P{)#ounkgEqFzJ+7{<0-}Y`u)p5!xE2RfopM|&6tElI z3=APi`D1!>4KniySW)B5p$7^$-xO;1+7Dy!0;3gcR9AjJW4DF5g~c6|zkmOZAM2HA z!(iI4cRYWoKfvsbw3O7Cg4*-vlwf&nL3g6vWX230dH1M=(EIg=xoEn(W zE@9HhsL|<_mzNie&J;JBz;I5*pQli|hx_~0Mysam#N7WSmosvZ-~hR(5a-rpfcCg% zs&wiGKjUc?zTc*su383H8*;8GYmX`$x|6J*o{`~p#E=qS-m$#VY z#MAB9UUFHN7Ure9G4W+f*X-#_ zFtI8>e%U@W^|A{)@Sg9^4e)_{flIw#km;W@y(wtW3loQA;L00l$j$Apgtg;fVId3* z(ik}e1?#6}Xx|~@#kN1=wXC7dix*}v1T)dNF+xVq!$?N==+W%o+zD)KY|KC#8ynkS zkZ-U8At*l%t(qz+MO$`G>*+v2nf(r{@v|K%3t{C&3H4D<3e<>haXrfhKQ zHCc{OgoTAsycx()nR(=q1%GVfTLIy5d$JYUFJrl03gb(HCth6(hNV%+3l<}JK4cAB zcsjq%2-C&}osh9nP{9Av(}CkYQt~(W7n4$Q)`tl&h^wiq2c7Q!&TVg}f!f*$?M$W1 zsv=x&iLk--_4TiI%{NdO;|Cn6*OyxkUR^A-6~H+=QmUra*4D1mG)!+sL`H6qZa$61 zlC!q9eypXHxK#y|-%S)SWPHq@vJ*u41d}0wdwVW0GJ^pIN};y)+b_AQ*- zlK>7DgYr7kzE#W=Ef6f@v?3zpC~6uSKQ`MwfBtN3Z$F;rh z;xcFZ{BNCAme)c3y|}ozvsN*L30*4b$dGB>?Og+V_a=kXlBSH7oOI>d{cFP?5@+l8 zCpX>4#>TQrOK|~+M!bF72M{X;kX3GBA*Q{(z31T{;;5*o9q0**b*dj(c)@Tz<-LO5fKr6a%d>g`EvLti#clI zAyA-{4h@tY7kfA;3-G2e%8mi5|LlL#r<}jPIG7=Cd>9FXIg#!~r$7dn=!t-V#jKnh zREWcW?nbn~lE?_XpGr&LB;hnDoEoohY9fbv(hdA0Ji;4?*VB-yvzwaqj+eqS%A=9D z<8+TYpnY9={;G8g9v%}P{}u|g)o3CjBG$IHs3;e}GSB|`v+e51n8%s)H2tZYN{9dD zLjYeGvd^5>hRKy@+r~Sei~jvcY5&6b5R?D*eY2Pf|0hqLaF^ajn9#%ic9e9{I599W zAxEM$JNRhG$p-*QR#jK4U#1u4FdNib-$_?tA_XR2_8Dd%IXQW0Y3Z2nwI67JRJ`Vx zKfZrIS?IhOmj{UzDyFdeKV7hWxxzRMNE2*^oTLhSFe6~@;=+65t8JYN=nuXZCsk{% zDllIUF{`AYfKXJ#US3{KiI4AITeFn5@GLAYHud&?2+1)VpL!6Yw}jHlv!olW62jnC zQ+_T?g@2e96@|_C-~kT{OC(I2ufcU5H+y6wFG&_5=6w(W<7MUs22{a8LDp5e$?x8M z0?cPC=(`d5pUTDy{vSx8djJs);I9q>#wv%)B{o^>SeuD*oXLo=wlP#;cq`|j~cNMwSE zdI4%7?AHJYrn`Xc|CFDjB&Vbdq^45L)*Ay;eEB0Z7(Wq81rP?L@5ww^=JDHXd{IB2}+SW%%cg1)3BUGON>+!${lwKir5G^JuN>dFh{5z)O zotmj=7Dq)+_^)CcOU-o;&-fQ3A`p~hXigtGM&JD*=raK8IsaMe)QF9 Z0Pnoe1vMT9^3P+TD9EbFluMZe{SUXqDQ5rx literal 0 HcmV?d00001 diff --git a/doc/source/data_structures.rst b/doc/source/data_structures.rst index 96fb953..0f72ba1 100644 --- a/doc/source/data_structures.rst +++ b/doc/source/data_structures.rst @@ -70,7 +70,7 @@ Basic Methods Relationship Tests ^^^^^^^^^^^^^^^^^^^ -* ``almost_equals(other)``: is shape almost the same as ``other`` (good when floating point precision issues make shapes slightly different) +* ``geom_almost_equals(other)``: is shape almost the same as ``other`` (good when floating point precision issues make shapes slightly different) * ``contains(other)``: is shape contained within ``other`` * ``intersects(other)``: does shape intersect ``other`` diff --git a/doc/source/geometric_manipulations.rst b/doc/source/geometric_manipulations.rst index 2996ee2..29f6613 100644 --- a/doc/source/geometric_manipulations.rst +++ b/doc/source/geometric_manipulations.rst @@ -1,92 +1,74 @@ Geometric Manipulations ======================== +*geopandas* makes available all the tools for geometric manipulations in the `*shapely* library `_. - - -Set-theoretic Methods -~~~~~~~~~~~~~~~~~~~~~ - -.. attribute:: GeoSeries.boundary - - Returns a ``GeoSeries`` of lower dimensional objects representing - each geometries's set-theoretic `boundary`. - -.. method:: GeoSeries.difference(other) - - Returns a ``GeoSeries`` of the points in each geometry that - are not in the *other* object. - -.. method:: GeoSeries.intersection(other) - - Returns a ``GeoSeries`` of the intersection of each object with the `other` - geometric object. - -.. method:: GeoSeries.symmetric_difference(other) - - Returns a ``GeoSeries`` of the points in each object not in the `other` - geometric object, and the points in the `other` not in this object. - -.. method:: GeoSeries.union(other) - - Returns a ``GeoSeries`` of the union of points from each object and the - `other` geometric object. - - -.. attribute:: GeoSeries.unary_union - - Return a geometry containing the union of all geometries in the ``GeoSeries``. - +Note that documentation for all set-theoretic tools for creating new shapes using the relationship between two different spatial datasets -- like creating intersections, or differences -- can be found on the :doc:`set operations ` page. Constructive Methods -~~~~~~~~~~~~~~~~~~~~~ +~~~~~~~~~~~~~~~~~~~~ -.. method:: GeoSeries.buffer(distance, resolution=16) +.. method:: GeoSeries.buffer(distance, resolution=16) Returns a ``GeoSeries`` of geometries representing all points within a given `distance` of each geometric object. -.. attribute:: GeoSeries.convex_hull +.. attribute:: GeoSeries.boundary + + Returns a ``GeoSeries`` of lower dimensional objects representing + each geometries's set-theoretic `boundary`. + +.. attribute:: GeoSeries.centroid + + Returns a ``GeoSeries`` of points for each geometric centroid. + +.. attribute:: GeoSeries.convex_hull Returns a ``GeoSeries`` of geometries representing the smallest convex `Polygon` containing all the points in each object unless the number of points in the object is less than three. For two points, the convex hull collapses to a `LineString`; for 1, a `Point`. -.. attribute:: GeoSeries.envelope +.. attribute:: GeoSeries.envelope Returns a ``GeoSeries`` of geometries representing the point or smallest rectangular polygon (with sides parallel to the coordinate axes) that contains each object. -.. method:: GeoSeries.simplify(tolerance, preserve_topology=True) +.. method:: GeoSeries.simplify(tolerance, preserve_topology=True) Returns a ``GeoSeries`` containing a simplified representation of each object. Affine transformations -~~~~~~~~~~~~~~~~~~~~~~~ +~~~~~~~~~~~~~~~~~~~~~~~~ -.. method:: GeoSeries.rotate(self, angle, origin='center', use_radians=False) +.. method:: GeoSeries.rotate(self, angle, origin='center', use_radians=False) Rotate the coordinates of the GeoSeries. -.. method:: GeoSeries.scale(self, xfact=1.0, yfact=1.0, zfact=1.0, origin='center') +.. method:: GeoSeries.scale(self, xfact=1.0, yfact=1.0, zfact=1.0, origin='center') Scale the geometries of the GeoSeries along each (x, y, z) dimensio. -.. method:: GeoSeries.skew(self, angle, origin='center', use_radians=False) +.. method:: GeoSeries.skew(self, angle, origin='center', use_radians=False) Shear/Skew the geometries of the GeoSeries by angles along x and y dimensions. -.. method:: GeoSeries.translate(self, angle, origin='center', use_radians=False) +.. method:: GeoSeries.translate(self, angle, origin='center', use_radians=False) Shift the coordinates of the GeoSeries. -`Aggregating methods` +Aggregation Methods +~~~~~~~~~~~~~~~~~~~~ +.. attribute:: GeoSeries.unary_union + + Return a geometry containing the union of all geometries in the ``GeoSeries``. +Examples of Geometric Manipulations +------------------------------------ .. sourcecode:: python diff --git a/doc/source/index.rst b/doc/source/index.rst index 17d8c81..b380c35 100644 --- a/doc/source/index.rst +++ b/doc/source/index.rst @@ -32,6 +32,7 @@ such as PostGIS. Making Maps Managing Projections Geometric Manipulations + Set Operations with overlay Merging Data Geocoding Reference to All Attributes and Methods diff --git a/doc/source/reference.rst b/doc/source/reference.rst index a05e840..2b95ab7 100644 --- a/doc/source/reference.rst +++ b/doc/source/reference.rst @@ -124,15 +124,6 @@ The following Shapely methods and attributes are available on `Set-theoretic Methods` -.. attribute:: GeoSeries.boundary - - Returns a ``GeoSeries`` of lower dimensional objects representing - each geometries's set-theoretic `boundary`. - -.. attribute:: GeoSeries.centroid - - Returns a ``GeoSeries`` of points for each geometric centroid. - .. method:: GeoSeries.difference(other) Returns a ``GeoSeries`` of the points in each geometry that @@ -160,6 +151,15 @@ The following Shapely methods and attributes are available on Returns a ``GeoSeries`` of geometries representing all points within a given `distance` of each geometric object. +.. attribute:: GeoSeries.boundary + + Returns a ``GeoSeries`` of lower dimensional objects representing + each geometries's set-theoretic `boundary`. + +.. attribute:: GeoSeries.centroid + + Returns a ``GeoSeries`` of points for each geometric centroid. + .. attribute:: GeoSeries.convex_hull Returns a ``GeoSeries`` of geometries representing the smallest diff --git a/doc/source/set_operations.rst b/doc/source/set_operations.rst new file mode 100644 index 0000000..bd964ba --- /dev/null +++ b/doc/source/set_operations.rst @@ -0,0 +1,76 @@ +.. ipython:: python + :suppress: + + import geopandas as gpd + world = gpd.GeoDataFrame().from_file('_example_data/naturalearth_lowres.shp') + capitals = gpd.GeoDataFrame().from_file('_example_data/naturalearth_cities.shp') + + # For spatial join + countries = world[['geometry', 'name']] + + # Project + countries = countries.to_crs('+init=epsg:3395')[countries.name!="Antarctica"] + capitals = capitals.to_crs('+init=epsg:3395') + + +Set-Operations with Overlay +============================ + +When working with multiple spatial datasets -- especially multiple *polygon* or *line* datasets -- users often wish to create new shapes based on places where those datasets overlap (or don't overlap). These manipulations are often referred using the language of sets -- intersections, unions, and differences. These types of operations are made available in the *geopandas* library through the ``overlay`` function. + +The basic idea is demonstrated by the graphic below but keep in mind that overlays operate at the DataFrame level, not on individual geometries, and the properties from both are retained. In effect, for every shape in the first GeoDataFrame, this operation is executed against every other shape in the other GeoDataFrame: + +.. image:: _static/overlay_operations.png + +**Source: QGIS Documentation** + +(Note to users familiar with the *shapely* library: ``overlay`` can be thought of as offering versions of the standard *shapely* set-operations that deal with the complexities of applying set operations to two *GeoSeries*. The standard *shapely* set-operations are also available as ``GeoSeries`` methods.) + + +Overlay Example +----------------- + +To illustrate the ``overlay`` function, consider the following case in which one wishes to identify the "core" portion of each country -- defined as areas within 500km of a capital -- using a ``GeoDataFrame`` of countries and a ``GeoDataFrame`` of capitals. + +.. ipython:: python + + # Look at countries: + @savefig world_basic.png width=5in + countries.plot(); + + # Now buffer cities to find area within 500km. + # Check CRS -- World Mercator, units of meters. + capitals.crs + + # make 500km buffer + capitals['geometry']= capitals.buffer(500000) + @savefig capital_buffers.png width=5in + capitals.plot(); + + +To select only the portion of countries within 500km of a capital, we specify the ``how`` option to be "intersect", which creates a new set of polygons where these two layers overlap: + +.. ipython:: python + + from geopandas.tools import overlay + country_cores = overlay(countries, capitals, how='intersection') + @savefig country_cores.png width=5in + country_cores.plot(); + +Changing the "how" option allows for different types of overlay operations. For example, if we were interested in the portions of countries *far* from capitals (the peripheries), we would compute the difference of the two. + +.. ipython:: python + + country_peripheries = overlay(countries, capitals, how='difference') + @savefig country_peripheries.png width=5in + country_peripheries.plot(); + +More Examples +----------------- + +A larger set of examples of the use of ``overlay`` can be found `here `_ + + + +.. toctree:: + :maxdepth: 2 From 6b799a21c1feb9bc7a17b781b3fdfcca687845ec Mon Sep 17 00:00:00 2001 From: Nick Eubank Date: Tue, 17 May 2016 15:40:30 -0700 Subject: [PATCH 10/34] Fixes 305 and 306 (#307) --- geopandas/tests/test_overlay.py | 22 ++++++++++++++++++++++ geopandas/tools/overlay.py | 11 ++++++++--- 2 files changed, 30 insertions(+), 3 deletions(-) diff --git a/geopandas/tests/test_overlay.py b/geopandas/tests/test_overlay.py index 9aa1e3c..051618a 100644 --- a/geopandas/tests/test_overlay.py +++ b/geopandas/tests/test_overlay.py @@ -86,6 +86,28 @@ class TestDataFrame(unittest.TestCase): df = overlay(self.polydf, polydf2r, how="union") self.assertTrue('Shape_Area_2' in df.columns and 'Shape_Area' in df.columns) + def test_geometry_not_named_geometry(self): + # Issue #306 + # Add points and flip names + polydf3 = self.polydf.copy() + polydf3 = polydf3.rename(columns={'geometry':'polygons'}) + polydf3 = polydf3.set_geometry('polygons') + polydf3['geometry'] = self.pointdf.geometry.loc[0:4] + self.assertTrue(polydf3.geometry.name == 'polygons') + + df = overlay(polydf3, self.polydf2, how="union") + self.assertTrue(type(df) is GeoDataFrame) + + df2 = overlay(self.polydf, self.polydf2, how="union") + self.assertTrue(df.geom_almost_equals(df2).all()) + + def test_geoseries_warning(self): + # Issue #305 + + def f(): + overlay(self.polydf, self.polydf2.geometry, how="union") + self.assertRaises(NotImplementedError, f) + diff --git a/geopandas/tools/overlay.py b/geopandas/tools/overlay.py index 4d14cd1..d19a13b 100644 --- a/geopandas/tools/overlay.py +++ b/geopandas/tools/overlay.py @@ -30,8 +30,10 @@ def _extract_rings(df): """ poly_msg = "overlay only takes GeoDataFrames with (multi)polygon geometries" rings = [] + geometry_column = df.geometry.name + for i, feat in df.iterrows(): - geom = feat.geometry + geom = feat[geometry_column] if geom.type not in ['Polygon', 'MultiPolygon']: raise TypeError(poly_msg) @@ -82,6 +84,9 @@ def overlay(df1, df2, how, use_sindex=True): raise ValueError("`how` was \"%s\" but is expected to be in %s" % \ (how, allowed_hows)) + if isinstance(df1, GeoSeries) or isinstance(df2, GeoSeries): + raise NotImplementedError("overlay currently only implemented for GeoDataFrames") + # Collect the interior and exterior rings rings1 = _extract_rings(df1) rings2 = _extract_rings(df2) @@ -125,13 +130,13 @@ def overlay(df1, df2, how, use_sindex=True): prop2 = None for cand_id in candidates1: cand = df1.ix[cand_id] - if cent.intersects(cand.geometry): + if cent.intersects(cand[df1.geometry.name]): df1_hit = True prop1 = cand break # Take the first hit for cand_id in candidates2: cand = df2.ix[cand_id] - if cent.intersects(cand.geometry): + if cent.intersects(cand[df2.geometry.name]): df2_hit = True prop2 = cand break # Take the first hit From 82dd425e179cb02eb2b245a60d1859bb55ce5a08 Mon Sep 17 00:00:00 2001 From: Joris Van den Bossche Date: Fri, 20 May 2016 01:05:45 +0200 Subject: [PATCH 11/34] DOC: loading example datasets with new method (#321) * API: provide datasets submodule in top-level namespace * DOC: loading example datasets with new method --- doc/source/data_structures.rst | 14 ++++++++--- doc/source/mapping.rst | 39 +++++++++++++++------------- doc/source/mergingdata.rst | 46 ++++++++++++++++++---------------- doc/source/projections.rst | 20 +++++++-------- doc/source/set_operations.rst | 23 ++++++++++------- geopandas/__init__.py | 2 ++ 6 files changed, 84 insertions(+), 60 deletions(-) diff --git a/doc/source/data_structures.rst b/doc/source/data_structures.rst index 0f72ba1..c80361d 100644 --- a/doc/source/data_structures.rst +++ b/doc/source/data_structures.rst @@ -4,8 +4,7 @@ :suppress: import geopandas as gpd - world = gpd.GeoDataFrame().from_file('_example_data/naturalearth_lowres.shp') - world = world.rename(columns={'geometry': 'borders'}).set_geometry('borders') + Data Structures ========================================= @@ -90,18 +89,27 @@ An example using the ``worlds`` GeoDataFrame: .. ipython:: python + world = gpd.read_file(gpd.datasets.get_path('naturalearth_lowres')) + world.head() #Plot countries @savefig world_borders.png width=3in world.plot(); -Currently, the column named "borders" with country borders is the active +Currently, the column named "geometry" with country borders is the active geometry column: .. ipython:: python world.geometry.name +We can also rename this column to "borders": + +.. ipython:: python + + world = world.rename(columns={'geometry': 'borders'}).set_geometry('borders') + world.geometry.name + Now, we create centroids and make it the geometry: .. ipython:: python diff --git a/doc/source/mapping.rst b/doc/source/mapping.rst index 2ce1cb6..ff75960 100644 --- a/doc/source/mapping.rst +++ b/doc/source/mapping.rst @@ -4,19 +4,25 @@ :suppress: import geopandas as gpd - world = gpd.GeoDataFrame().from_file('_example_data/naturalearth_lowres.shp') - cities = gpd.GeoDataFrame().from_file('_example_data/naturalearth_cities.shp') - Mapping Tools ========================================= -*geopandas* provides a high-level interface to the ``matplotlib`` library for making maps. Mapping shapes is as easy as using the ``plot()`` method on a ``GeoSeries`` or ``GeoDataFrame``. +*geopandas* provides a high-level interface to the ``matplotlib`` library for making maps. Mapping shapes is as easy as using the ``plot()`` method on a ``GeoSeries`` or ``GeoDataFrame``. + +Loading some example data: .. ipython:: python - + + world = gpd.read_file(gpd.datasets.get_path('naturalearth_lowres')) + cities = gpd.read_file(gpd.datasets.get_path('naturalearth_cities')) + +We can now plot those GeoDataFrames: + +.. ipython:: python + # Examine country GeoDataFrame world.head() @@ -24,13 +30,13 @@ Mapping Tools @savefig world_randomcolors.png width=5in world.plot(); -Note that in general, any options one can pass to `pyplot `_ in ``matplotlib`` (or `style options that work for lines `_) can be passed to the ``plot()`` method. +Note that in general, any options one can pass to `pyplot `_ in ``matplotlib`` (or `style options that work for lines `_) can be passed to the ``plot()`` method. Chloropleth Maps ----------------- -*geopandas* makes it easy to create Chloropleth maps (maps where the color of each shape is based on the value of an associated variable). Simply use the plot command with the ``column`` argument set to the column whose values you want used to assign colors. +*geopandas* makes it easy to create Chloropleth maps (maps where the color of each shape is based on the value of an associated variable). Simply use the plot command with the ``column`` argument set to the column whose values you want used to assign colors. .. ipython:: python @@ -52,7 +58,7 @@ One can also modify the colors used by ``plot`` with the ``cmap`` option (for a world.plot(column='gdp_per_cap', cmap='OrRd'); -The way color maps are scaled can also be manipulated with the ``scheme`` option (if you have ``pysal`` installed, which can be accomplished via ``conda install pysal``). By default, ``scheme`` is set to 'equal_intervals', but it can also be adjusted to any other `pysal option `_, like 'quantiles', 'percentiles', etc. +The way color maps are scaled can also be manipulated with the ``scheme`` option (if you have ``pysal`` installed, which can be accomplished via ``conda install pysal``). By default, ``scheme`` is set to 'equal_intervals', but it can also be adjusted to any other `pysal option `_, like 'quantiles', 'percentiles', etc. .. ipython:: python @@ -63,14 +69,14 @@ The way color maps are scaled can also be manipulated with the ``scheme`` option Maps with Layers ----------------- -There are two strategies for making a map with multiple layers -- one more succinct, and one that is a littel more flexible. +There are two strategies for making a map with multiple layers -- one more succinct, and one that is a littel more flexible. -Before combining maps, however, remember to always ensure they share a common CRS (so they will align). +Before combining maps, however, remember to always ensure they share a common CRS (so they will align). .. ipython:: python - + # Look at capitals - # Note use of standard `pyplot` line style options + # Note use of standard `pyplot` line style options @savefig capitals.png width=5in cities.plot(marker='*', color='green', markersize=5); @@ -96,10 +102,10 @@ Before combining maps, however, remember to always ensure they share a common CR import matplotlib.pyplot as plt fig, ax = plt.subplots() - # set aspect to equal. This is done automatically - # when using *geopandas* plot on it's own, but not when - # working with pyplot directly. - ax.set_aspect('equal') + # set aspect to equal. This is done automatically + # when using *geopandas* plot on it's own, but not when + # working with pyplot directly. + ax.set_aspect('equal') world.plot(ax=ax, color='white') cities.plot(ax=ax, marker='o', color='red', markersize=5) @@ -112,4 +118,3 @@ Other Resources Links to jupyter Notebooks for different mapping tasks: `Making Heat Maps `_ - diff --git a/doc/source/mergingdata.rst b/doc/source/mergingdata.rst index 9110ac1..ce75ff2 100644 --- a/doc/source/mergingdata.rst +++ b/doc/source/mergingdata.rst @@ -4,17 +4,6 @@ :suppress: import geopandas as gpd - world = gpd.GeoDataFrame().from_file('_example_data/naturalearth_lowres.shp') - cities = gpd.GeoDataFrame().from_file('_example_data/naturalearth_cities.shp') - - # For attribute join - country_shapes = world[['geometry', 'iso_a3']] - country_names = world[['name', 'iso_a3']] - - # For spatial join - countries = world[['geometry', 'name']] - countries = countries.rename(columns={'name':'country'}) - Merging Data @@ -24,18 +13,34 @@ There are two ways to combine datasets in *geopandas* -- attribute joins and spa In an attribute join, a ``GeoSeries`` or ``GeoDataFrame`` is combined with a regular *pandas* ``Series`` or ``DataFrame`` based on a common variable. This is analogous to normal merging or joining in *pandas*. -In a Spatial Join, observations from to ``GeoSeries`` or ``GeoDataFrames`` are combined based on their spatial relationship to one another. +In a Spatial Join, observations from to ``GeoSeries`` or ``GeoDataFrames`` are combined based on their spatial relationship to one another. + +In the following examples, we use these datasets: + +.. ipython:: python + + world = gpd.read_file(gpd.datasets.get_path('naturalearth_lowres')) + cities = gpd.read_file(gpd.datasets.get_path('naturalearth_cities')) + + # For attribute join + country_shapes = world[['geometry', 'iso_a3']] + country_names = world[['name', 'iso_a3']] + + # For spatial join + countries = world[['geometry', 'name']] + countries = countries.rename(columns={'name':'country'}) + Attribute Joins ---------------- -Attribute joins are accomplished using the ``merge`` method. In general, it is recommended to use the ``merge`` method called from the spatial dataset. With that said, the stand-alone ``merge`` function will work if the GeoDataFrame is in the ``left`` argument; if a DataFrame is in the ``left`` argument and a GeoDataFrame is in the ``right`` position, the result will no longer be a GeoDataFrame. +Attribute joins are accomplished using the ``merge`` method. In general, it is recommended to use the ``merge`` method called from the spatial dataset. With that said, the stand-alone ``merge`` function will work if the GeoDataFrame is in the ``left`` argument; if a DataFrame is in the ``left`` argument and a GeoDataFrame is in the ``right`` position, the result will no longer be a GeoDataFrame. For example, consider the following merge that adds full names to a ``GeoDataFrame`` that initially has only ISO codes for each country by merging it with a *pandas* ``DataFrame``. -.. ipython:: python - +.. ipython:: python + # `country_shapes` is GeoDataFrame with country shapes and iso codes country_shapes.head() @@ -51,13 +56,13 @@ For example, consider the following merge that adds full names to a ``GeoDataFra Spatial Joins ---------------- -In a Spatial Join, two geometry objects are merged based on their spatial relationship to one another. +In a Spatial Join, two geometry objects are merged based on their spatial relationship to one another. .. ipython:: python - # One GeoDataFrame of countries, one of Cities. - # Want to merge so we can get each city's country. + # One GeoDataFrame of countries, one of Cities. + # Want to merge so we can get each city's country. countries.head() cities.head() @@ -67,7 +72,6 @@ In a Spatial Join, two geometry objects are merged based on their spatial relati cities_with_country.head() -The ``op`` options determines the type of join operation to apply. ``op`` can be set to "intersects", "within" or "contains" (these are all equivalent when joining points to polygons, but differ when joining polygons to other polygons or lines). - -Note more complicated spatial relationships can be studied by combining geometric operations with spatial join. To find all polygons within a given distance of a point, for example, one can first use the ``buffer`` method to expand each point into a circle of appropriate radius, then intersect those buffered circles with the polygons in question. +The ``op`` options determines the type of join operation to apply. ``op`` can be set to "intersects", "within" or "contains" (these are all equivalent when joining points to polygons, but differ when joining polygons to other polygons or lines). +Note more complicated spatial relationships can be studied by combining geometric operations with spatial join. To find all polygons within a given distance of a point, for example, one can first use the ``buffer`` method to expand each point into a circle of appropriate radius, then intersect those buffered circles with the polygons in question. diff --git a/doc/source/projections.rst b/doc/source/projections.rst index de4f03c..bfca83c 100644 --- a/doc/source/projections.rst +++ b/doc/source/projections.rst @@ -4,8 +4,6 @@ :suppress: import geopandas as gpd - world = gpd.GeoDataFrame().from_file('_example_data/naturalearth_lowres.shp') - Managing Projections @@ -18,11 +16,11 @@ Coordinate Reference Systems CRS are important because the geometric shapes in a GeoSeries or GeoDataFrame object are simply a collection of coordinates in an arbitrary space. A CRS tells Python how those coordinates related to places on the Earth. -CRS are referred to using codes called `proj4 strings `_. You can find the codes for most commonly used projections from `www.spatialreference.org `_ or `remotesensing.org `_. +CRS are referred to using codes called `proj4 strings `_. You can find the codes for most commonly used projections from `www.spatialreference.org `_ or `remotesensing.org `_. -The same CRS can often be referred to in many ways. For example, one of the most commonly used CRS is the WGS84 latitude-longitude projection. One `proj4` representation of this projection is: ``"+proj=longlat +ellps=WGS84 +datum=WGS84 +no_defs"``. But common projections can also be referred to by `EPSG` codes, so this same projection can also called using the `proj4` string ``"+init=epsg:4326"``. +The same CRS can often be referred to in many ways. For example, one of the most commonly used CRS is the WGS84 latitude-longitude projection. One `proj4` representation of this projection is: ``"+proj=longlat +ellps=WGS84 +datum=WGS84 +no_defs"``. But common projections can also be referred to by `EPSG` codes, so this same projection can also called using the `proj4` string ``"+init=epsg:4326"``. -*geopandas* can accept lots of representations of CRS, including the `proj4` string itself (``"+proj=longlat +ellps=WGS84 +datum=WGS84 +no_defs"``) or parameters broken out in a dictionary: ``{'proj': 'latlong', 'ellps': 'WGS84', 'datum': 'WGS84', 'no_defs': True}``). In addition, some functions will take `EPSG` codes directly. +*geopandas* can accept lots of representations of CRS, including the `proj4` string itself (``"+proj=longlat +ellps=WGS84 +datum=WGS84 +no_defs"``) or parameters broken out in a dictionary: ``{'proj': 'latlong', 'ellps': 'WGS84', 'datum': 'WGS84', 'no_defs': True}``). In addition, some functions will take `EPSG` codes directly. For reference, a few very common projections and their proj4 strings: @@ -33,18 +31,18 @@ For reference, a few very common projections and their proj4 strings: Setting a Projection ---------------------- -There are two relevant operations for projections: setting a projection and re-projecting. +There are two relevant operations for projections: setting a projection and re-projecting. Setting a projection may be necessary when for some reason *geopandas* has coordinate data (x-y values), but no information about how those coordinates refer to locations in the real world. Setting a projection is how one tells *geopandas* how to interpret coordinates. If no CRS is set, *geopandas* geometry operations will still work, but coordinate transformations will not be possible and exported files may not be interpreted correctly by other software. -Be aware that **most of the time** you don't have to set a projection. Data loaded from a reputable source (using the ``from_file()`` command) *should* always include projection information. You can see an objects current CRS through the ``crs`` attribute: ``my_geoseries.crs``. +Be aware that **most of the time** you don't have to set a projection. Data loaded from a reputable source (using the ``from_file()`` command) *should* always include projection information. You can see an objects current CRS through the ``crs`` attribute: ``my_geoseries.crs``. From time to time, however, you may get data that does not include a projection. In this situation, you have to set the CRS so *geopandas* knows how to interpret the coordinates. For example, if you convert a spreadsheet of latitudes and longitudes into a GeoSeries by hand, you would set the projection by assigning the WGS84 latitude-longitude CRS to the ``crs`` attribute: .. sourcecode:: python - + my_geoseries.crs = {'init' :'epsg:4326'} @@ -54,7 +52,10 @@ Re-Projecting Re-projecting is the process of changing the representation of locations from one coordinate system to another. All projections of locations on the Earth into a two-dimensional plane `are distortions `_, the projection that is best for your application may be different from the projection associated with the data you import. In these cases, data can be re-projected using the ``to_crs`` command: .. ipython:: python - + + # load example data + world = gpd.read_file(gpd.datasets.get_path('naturalearth_lowres')) + # Check original projection # (it's Platte Carre! x-y are long and lat) world.crs @@ -68,4 +69,3 @@ Re-projecting is the process of changing the representation of locations from on world = world.to_crs({'init': 'epsg:3395'}) # world.to_crs(epsg=3395) would also work @savefig world_reproj.png width=3in world.plot(); - diff --git a/doc/source/set_operations.rst b/doc/source/set_operations.rst index bd964ba..2ffc57c 100644 --- a/doc/source/set_operations.rst +++ b/doc/source/set_operations.rst @@ -2,15 +2,6 @@ :suppress: import geopandas as gpd - world = gpd.GeoDataFrame().from_file('_example_data/naturalearth_lowres.shp') - capitals = gpd.GeoDataFrame().from_file('_example_data/naturalearth_cities.shp') - - # For spatial join - countries = world[['geometry', 'name']] - - # Project - countries = countries.to_crs('+init=epsg:3395')[countries.name!="Antarctica"] - capitals = capitals.to_crs('+init=epsg:3395') Set-Operations with Overlay @@ -30,6 +21,20 @@ The basic idea is demonstrated by the graphic below but keep in mind that overla Overlay Example ----------------- +First, we load some example data: + +.. ipython:: python + + world = gpd.read_file(gpd.datasets.get_path('naturalearth_lowres')) + capitals = gpd.read_file(gpd.datasets.get_path('naturalearth_cities')) + + # Select some columns + countries = world[['geometry', 'name']] + + # Project to crs that uses meters as distance measure + countries = countries.to_crs('+init=epsg:3395')[countries.name!="Antarctica"] + capitals = capitals.to_crs('+init=epsg:3395') + To illustrate the ``overlay`` function, consider the following case in which one wishes to identify the "core" portion of each country -- defined as areas within 500km of a capital -- using a ``GeoDataFrame`` of countries and a ``GeoDataFrame`` of capitals. .. ipython:: python diff --git a/geopandas/__init__.py b/geopandas/__init__.py index 032ec42..3de572a 100644 --- a/geopandas/__init__.py +++ b/geopandas/__init__.py @@ -11,6 +11,8 @@ from geopandas.io.sql import read_postgis from geopandas.tools import sjoin from geopandas.tools import overlay +import geopandas.datasets + # make the interactive namespace easier to use # for `from geopandas import *` demos. import geopandas as gpd From 320b75d18539a5fe6d431211050d765c4127c6ce Mon Sep 17 00:00:00 2001 From: Joris Van den Bossche Date: Fri, 20 May 2016 01:06:12 +0200 Subject: [PATCH 12/34] DOC: update minimum pandas version in install docs (#319) --- doc/source/install.rst | 6 ++---- 1 file changed, 2 insertions(+), 4 deletions(-) diff --git a/doc/source/install.rst b/doc/source/install.rst index 4153f5a..37d80e7 100644 --- a/doc/source/install.rst +++ b/doc/source/install.rst @@ -20,9 +20,7 @@ You may install the latest development version by cloning the pip install . It is also possible to install the latest development version -available on PyPI with `pip` by adding the ``--pre`` flag for pip 1.4 -and later, or to use `pip` to install directly from the GitHub -repository with:: +directly from the GitHub repository with:: pip install git+git://github.com/geopandas/geopandas.git @@ -32,7 +30,7 @@ Dependencies Installation via `conda` should also install all dependencies, but a complete list is as follows: - `numpy`_ -- `pandas`_ (version 0.13 or later) +- `pandas`_ (version 0.15.2 or later) - `shapely`_ - `fiona`_ - `six`_ From 0f82cedf5540373ca39a87f436d47c846ab0bcb5 Mon Sep 17 00:00:00 2001 From: Joris Van den Bossche Date: Thu, 26 May 2016 23:28:07 +0200 Subject: [PATCH 13/34] BUG: preserve metadata on merge/concat (#247, #320) (#322) --- geopandas/geodataframe.py | 22 +++++++++---- geopandas/tests/test_merge.py | 62 +++++++++++++++++++++++++++++++++++ 2 files changed, 77 insertions(+), 7 deletions(-) create mode 100644 geopandas/tests/test_merge.py diff --git a/geopandas/geodataframe.py b/geopandas/geodataframe.py index ef6e071..12e24b9 100644 --- a/geopandas/geodataframe.py +++ b/geopandas/geodataframe.py @@ -409,10 +409,18 @@ class GeoDataFrame(GeoPandasBase, DataFrame): return GeoDataFrame def __finalize__(self, other, method=None, **kwargs): - """ propagate metadata from other to self """ - # NOTE: backported from pandas master (upcoming v0.13) - for name in self._metadata: - object.__setattr__(self, name, getattr(other, name, None)) + """propagate metadata from other to self """ + # merge operation: using metadata of the left object + if method == 'merge': + for name in self._metadata: + object.__setattr__(self, name, getattr(other.left, name, None)) + # concat operation: using metadata of the first object + elif method == 'concat': + for name in self._metadata: + object.__setattr__(self, name, getattr(other.objs[0], name, None)) + else: + for name in self._metadata: + object.__setattr__(self, name, getattr(other, name, None)) return self def copy(self, deep=True): @@ -443,7 +451,7 @@ class GeoDataFrame(GeoPandasBase, DataFrame): def dissolve(self, by=None, aggfunc='first'): """ - Dissolve geometries within `groupby` into single observation. + Dissolve geometries within `groupby` into single observation. Parameters ---------- @@ -473,14 +481,14 @@ class GeoDataFrame(GeoPandasBase, DataFrame): merged_geom = block.unary_union new_index = block.drop(self.geometry.name, axis=1).iloc[0][by] - merged_w_index = GeoSeries(merged_geom, index=Index(Series(new_index),name=by), + merged_w_index = GeoSeries(merged_geom, index=Index(Series(new_index),name=by), name=self.geometry.name) return merged_w_index g = geometry.groupby(by=by, group_keys=False).apply(merge_geometries) - aggregated_geometry = GeoDataFrame(g, + aggregated_geometry = GeoDataFrame(g, index=g.index, geometry=self.geometry.name) # Recombine diff --git a/geopandas/tests/test_merge.py b/geopandas/tests/test_merge.py new file mode 100644 index 0000000..17cd238 --- /dev/null +++ b/geopandas/tests/test_merge.py @@ -0,0 +1,62 @@ +from __future__ import absolute_import + +import pandas as pd +from shapely.geometry import Point + +from geopandas import GeoDataFrame, GeoSeries +from geopandas.tests.util import unittest + + +class TestMerging(unittest.TestCase): + + def setUp(self): + + self.gseries = GeoSeries([Point(i, i) for i in range(3)]) + self.series = pd.Series([1, 2, 3]) + self.gdf = GeoDataFrame({'geometry': self.gseries, 'values': range(3)}) + self.df = pd.DataFrame({'col1': [1, 2, 3], 'col2': [0.1, 0.2, 0.3]}) + + def _check_metadata(self, gdf, geometry_column_name='geometry', crs=None): + + self.assertEqual(gdf._geometry_column_name, geometry_column_name) + self.assertEqual(gdf.crs, crs) + + def test_merge(self): + + res = self.gdf.merge(self.df, left_on='values', right_on='col1') + + # check result is a GeoDataFrame + self.assert_(isinstance(res, GeoDataFrame)) + + # check geometry property gives GeoSeries + self.assert_(isinstance(res.geometry, GeoSeries)) + + # check metadata + self._check_metadata(res) + + ## test that crs and other geometry name are preserved + self.gdf.crs = {'init' :'epsg:4326'} + self.gdf = (self.gdf.rename(columns={'geometry': 'points'}) + .set_geometry('points')) + res = self.gdf.merge(self.df, left_on='values', right_on='col1') + self.assert_(isinstance(res, GeoDataFrame)) + self.assert_(isinstance(res.geometry, GeoSeries)) + self._check_metadata(res, 'points', self.gdf.crs) + + def test_concat_axis0(self): + + res = pd.concat([self.gdf, self.gdf]) + + self.assertEqual(res.shape, (6, 2)) + self.assert_(isinstance(res, GeoDataFrame)) + self.assert_(isinstance(res.geometry, GeoSeries)) + self._check_metadata(res) + + def test_concat_axis1(self): + + res = pd.concat([self.gdf, self.df], axis=1) + + self.assertEqual(res.shape, (3, 4)) + self.assert_(isinstance(res, GeoDataFrame)) + self.assert_(isinstance(res.geometry, GeoSeries)) + self._check_metadata(res) From f58b5f46ade33a59f25bebf0523a4cfb716cd540 Mon Sep 17 00:00:00 2001 From: Nick Eubank Date: Fri, 27 May 2016 13:04:40 -0700 Subject: [PATCH 14/34] dissolve bug fixes (#323) --- geopandas/geodataframe.py | 36 ++++++++++---------- {tests => geopandas/tests}/test_dissolve.py | 37 ++++++++++++++++----- 2 files changed, 46 insertions(+), 27 deletions(-) rename {tests => geopandas/tests}/test_dissolve.py (56%) diff --git a/geopandas/geodataframe.py b/geopandas/geodataframe.py index 12e24b9..8c9e5b3 100644 --- a/geopandas/geodataframe.py +++ b/geopandas/geodataframe.py @@ -449,9 +449,14 @@ class GeoDataFrame(GeoPandasBase, DataFrame): plot.__doc__ = plot_dataframe.__doc__ - def dissolve(self, by=None, aggfunc='first'): + def dissolve(self, by=None, aggfunc='first', as_index=True): """ Dissolve geometries within `groupby` into single observation. + This is accomplished by applying the `unary_union` method + to all geometries within a groupself. + + Observations associated with each `groupby` group will be aggregated + using the `aggfunc`. Parameters ---------- @@ -460,6 +465,8 @@ class GeoDataFrame(GeoPandasBase, DataFrame): aggfunc : function or string, default "first" Aggregation function for manipulation of data associated with each group. Passed to pandas `groupby.agg` method. + as_index : boolean, default True + If true, groupby columns become index of result. Returns ------- @@ -467,33 +474,26 @@ class GeoDataFrame(GeoPandasBase, DataFrame): """ # Process non-spatial component - data = self.drop(labels=self.geometry.name, axis=1).copy() + data = self.drop(labels=self.geometry.name, axis=1) aggregated_data = data.groupby(by=by).agg(aggfunc) # Process spatial component - groupby_plus_geometry_cols = [self.geometry.name] - groupby_plus_geometry_cols.append(by) - geometry = self[groupby_plus_geometry_cols].copy() - def merge_geometries(block): - merged_geom = block.unary_union + return merged_geom - new_index = block.drop(self.geometry.name, axis=1).iloc[0][by] - merged_w_index = GeoSeries(merged_geom, index=Index(Series(new_index),name=by), - name=self.geometry.name) - return merged_w_index + g = self.groupby(by=by, group_keys=False)[self.geometry.name].agg(merge_geometries) - - g = geometry.groupby(by=by, group_keys=False).apply(merge_geometries) - - aggregated_geometry = GeoDataFrame(g, - index=g.index, - geometry=self.geometry.name) + # Aggregate + aggregated_geometry = GeoDataFrame(g, geometry=self.geometry.name) # Recombine aggregated = aggregated_geometry.join(aggregated_data) - aggregated = aggregated.set_geometry(self.geometry.name) + + # Reset if requested + if not as_index: + aggregated = aggregated.reset_index() + return aggregated def _dataframe_set_geometry(self, col, drop=False, inplace=False, crs=None): diff --git a/tests/test_dissolve.py b/geopandas/tests/test_dissolve.py similarity index 56% rename from tests/test_dissolve.py rename to geopandas/tests/test_dissolve.py index 58c91ac..45cc877 100644 --- a/tests/test_dissolve.py +++ b/geopandas/tests/test_dissolve.py @@ -8,6 +8,11 @@ from geopandas.tools import overlay from .util import unittest, download_nybb from pandas.util.testing import assert_frame_equal from pandas import Index +from distutils.version import LooseVersion +import pandas as pd + +pandas_0_15_problem = 'fails under pandas < 0.16 due to issue 324,'\ + 'not problem with dissolve.' class TestDataFrame(unittest.TestCase): @@ -28,7 +33,7 @@ class TestDataFrame(unittest.TestCase): others = self.polydf.loc[0:2,] collapsed = [others.geometry.unary_union, manhattan_bronx.geometry.unary_union] - merged_shapes = GeoDataFrame({'myshapes': collapsed}, geometry='myshapes', + merged_shapes = GeoDataFrame({'myshapes': collapsed}, geometry='myshapes', index=Index([5,6], name='manhattan_bronx')) # Different expected results @@ -40,25 +45,39 @@ class TestDataFrame(unittest.TestCase): self.mean['BoroCode'] = [4,1.5] + @unittest.skipIf(str(pd.__version__) < LooseVersion('0.16'), pandas_0_15_problem) def test_geom_dissolve(self): test = self.polydf.dissolve('manhattan_bronx') self.assertTrue(test.geometry.name == 'myshapes') self.assertTrue(test.geom_almost_equals(self.first).all()) + @unittest.skipIf(str(pd.__version__) < LooseVersion('0.16'), pandas_0_15_problem) def test_first_dissolve(self): test = self.polydf.dissolve('manhattan_bronx') - test = test.drop('myshapes', axis=1) - first = self.first.drop('myshapes', axis=1) - assert_frame_equal(first, test) + assert_frame_equal(self.first, test, check_column_type=False) + @unittest.skipIf(str(pd.__version__) < LooseVersion('0.16'), pandas_0_15_problem) def test_mean_dissolve(self): test = self.polydf.dissolve('manhattan_bronx', aggfunc='mean') - test = test.drop('myshapes', axis=1) - mean = self.mean.drop('myshapes', axis=1) - assert_frame_equal(mean, test) + assert_frame_equal(self.mean, test, check_column_type=False) test = self.polydf.dissolve('manhattan_bronx', aggfunc=np.mean) - test = test.drop('myshapes', axis=1) - assert_frame_equal(mean, test) + assert_frame_equal(self.mean, test, check_column_type=False) + @unittest.skipIf(str(pd.__version__) < LooseVersion('0.16'), pandas_0_15_problem) + def test_multicolumn_dissolve(self): + multi = self.polydf.copy() + multi['dup_col'] = multi.manhattan_bronx + multi_test = multi.dissolve(['manhattan_bronx', 'dup_col'], aggfunc='first') + first = self.first.copy() + first['dup_col'] = first.index + first = first.set_index([first.index, 'dup_col']) + + assert_frame_equal(multi_test, first, check_column_type=False) + + @unittest.skipIf(str(pd.__version__) < LooseVersion('0.16'), pandas_0_15_problem) + def test_reset_index(self): + test = self.polydf.dissolve('manhattan_bronx', as_index=False) + comparison = self.first.reset_index() + assert_frame_equal(comparison, test, check_column_type=False) From 659858d91199b076ed3c04367443c851b3996f89 Mon Sep 17 00:00:00 2001 From: Joris Van den Bossche Date: Thu, 28 Apr 2016 19:42:01 +0200 Subject: [PATCH 15/34] Add changelog for v0.2.0 --- CHANGELOG | 32 ++++++++++++++++++++++++++++++++ 1 file changed, 32 insertions(+) create mode 100644 CHANGELOG diff --git a/CHANGELOG b/CHANGELOG new file mode 100644 index 0000000..a7476bb --- /dev/null +++ b/CHANGELOG @@ -0,0 +1,32 @@ +Version 0.2.0 +------------- + +Improvements: + +* Complete overhaul of the documentation +* Addition of ``overlay`` to perform spatial overlays with polygons (#142) +* Addition of ``sjoin`` to perform spatial joins (#115, #145, #188) +* Addition of ``__geo_interface__`` that returns a python data structure + to represent the ``GeoSeries`` as a GeoJSON-like ``FeatureCollection`` (#116) + and ``iterfeatures`` method (#178) +* Addition of the ``explode`` (#146) and ``dissolve`` (#310, #311) methods. +* Improvements to plotting: ability to specify edge colors (#173), support for + the ``vmin``, ``vmax``, ``figsize``, ``linewidth`` keywords (#207), legends + for chloropleth plots (#210), color points by specifying a colormap (#186) or + a single color (#238). +* Larger flexibility of ``to_crs``, accepting both dicts and proj strings (#289) +* Addition of embedded example data, accessible through + ``geopandas.datasets.get_path``. + +API changes: + +* In the ``plot`` method, the ``axes`` keyword is renamed to ``ax`` for + consistency with pandas, and the ``colormap`` keyword is renamed to ``cmap`` + for consistency with matplotlib (#208, #228, #240). + +Bug fixes: + +* Properly handle rows with missing geometries (#139, #193). +* Fix ``GeoSeries.to_json`` (#263). +* Correctly serialize metadata when pickling (#199, #206). +* Fix ``merge`` and ``concat`` to return correct GeoDataFrame (#247, #320, #322). From dd815ac9d10eda490e520cc4e7228fd0873c5978 Mon Sep 17 00:00:00 2001 From: Joris Van den Bossche Date: Tue, 31 May 2016 11:27:53 +0200 Subject: [PATCH 16/34] MNT/BLD: use versioneer --- .gitattributes | 1 + MANIFEST.in | 2 + geopandas/__init__.py | 9 +- geopandas/_version.py | 484 +++++++++++ setup.cfg | 8 + setup.py | 54 +- versioneer.py | 1774 +++++++++++++++++++++++++++++++++++++++++ 7 files changed, 2276 insertions(+), 56 deletions(-) create mode 100644 .gitattributes create mode 100644 MANIFEST.in create mode 100644 geopandas/_version.py create mode 100644 versioneer.py diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..8340bd6 --- /dev/null +++ b/.gitattributes @@ -0,0 +1 @@ +geopandas/_version.py export-subst diff --git a/MANIFEST.in b/MANIFEST.in new file mode 100644 index 0000000..d7934e2 --- /dev/null +++ b/MANIFEST.in @@ -0,0 +1,2 @@ +include versioneer.py +include geopandas/_version.py diff --git a/geopandas/__init__.py b/geopandas/__init__.py index 3de572a..77eb04b 100644 --- a/geopandas/__init__.py +++ b/geopandas/__init__.py @@ -1,8 +1,3 @@ -try: - from geopandas.version import version as __version__ -except ImportError: - __version__ = '0.2.0.dev-unknown' - from geopandas.geoseries import GeoSeries from geopandas.geodataframe import GeoDataFrame @@ -18,3 +13,7 @@ import geopandas.datasets import geopandas as gpd import pandas as pd import numpy as np + +from ._version import get_versions +__version__ = get_versions()['version'] +del get_versions diff --git a/geopandas/_version.py b/geopandas/_version.py new file mode 100644 index 0000000..4604e70 --- /dev/null +++ b/geopandas/_version.py @@ -0,0 +1,484 @@ + +# This file helps to compute a version number in source trees obtained from +# git-archive tarball (such as those provided by githubs download-from-tag +# feature). Distribution tarballs (built by setup.py sdist) and build +# directories (produced by setup.py build) will contain a much shorter file +# that just contains the computed version number. + +# This file is released into the public domain. Generated by +# versioneer-0.16 (https://github.com/warner/python-versioneer) + +"""Git implementation of _version.py.""" + +import errno +import os +import re +import subprocess +import sys + + +def get_keywords(): + """Get the keywords needed to look up the version information.""" + # these strings will be replaced by git during git-archive. + # setup.py/versioneer.py will grep for the variable names, so they must + # each be defined on a line of their own. _version.py will just call + # get_keywords(). + git_refnames = "$Format:%d$" + git_full = "$Format:%H$" + keywords = {"refnames": git_refnames, "full": git_full} + return keywords + + +class VersioneerConfig: + """Container for Versioneer configuration parameters.""" + + +def get_config(): + """Create, populate and return the VersioneerConfig() object.""" + # these strings are filled in when 'setup.py versioneer' creates + # _version.py + cfg = VersioneerConfig() + cfg.VCS = "git" + cfg.style = "pep440" + cfg.tag_prefix = "v" + cfg.parentdir_prefix = "geopandas-" + cfg.versionfile_source = "geopandas/_version.py" + cfg.verbose = False + return cfg + + +class NotThisMethod(Exception): + """Exception raised if a method is not valid for the current scenario.""" + + +LONG_VERSION_PY = {} +HANDLERS = {} + + +def register_vcs_handler(vcs, method): # decorator + """Decorator to mark a method as the handler for a particular VCS.""" + def decorate(f): + """Store f in HANDLERS[vcs][method].""" + if vcs not in HANDLERS: + HANDLERS[vcs] = {} + HANDLERS[vcs][method] = f + return f + return decorate + + +def run_command(commands, args, cwd=None, verbose=False, hide_stderr=False): + """Call the given command(s).""" + assert isinstance(commands, list) + p = None + for c in commands: + try: + dispcmd = str([c] + args) + # remember shell=False, so use git.cmd on windows, not just git + p = subprocess.Popen([c] + args, cwd=cwd, stdout=subprocess.PIPE, + stderr=(subprocess.PIPE if hide_stderr + else None)) + break + except EnvironmentError: + e = sys.exc_info()[1] + if e.errno == errno.ENOENT: + continue + if verbose: + print("unable to run %s" % dispcmd) + print(e) + return None + else: + if verbose: + print("unable to find command, tried %s" % (commands,)) + return None + stdout = p.communicate()[0].strip() + if sys.version_info[0] >= 3: + stdout = stdout.decode() + if p.returncode != 0: + if verbose: + print("unable to run %s (error)" % dispcmd) + return None + return stdout + + +def versions_from_parentdir(parentdir_prefix, root, verbose): + """Try to determine the version from the parent directory name. + + Source tarballs conventionally unpack into a directory that includes + both the project name and a version string. + """ + dirname = os.path.basename(root) + if not dirname.startswith(parentdir_prefix): + if verbose: + print("guessing rootdir is '%s', but '%s' doesn't start with " + "prefix '%s'" % (root, dirname, parentdir_prefix)) + raise NotThisMethod("rootdir doesn't start with parentdir_prefix") + return {"version": dirname[len(parentdir_prefix):], + "full-revisionid": None, + "dirty": False, "error": None} + + +@register_vcs_handler("git", "get_keywords") +def git_get_keywords(versionfile_abs): + """Extract version information from the given file.""" + # the code embedded in _version.py can just fetch the value of these + # keywords. When used from setup.py, we don't want to import _version.py, + # so we do it with a regexp instead. This function is not used from + # _version.py. + keywords = {} + try: + f = open(versionfile_abs, "r") + for line in f.readlines(): + if line.strip().startswith("git_refnames ="): + mo = re.search(r'=\s*"(.*)"', line) + if mo: + keywords["refnames"] = mo.group(1) + if line.strip().startswith("git_full ="): + mo = re.search(r'=\s*"(.*)"', line) + if mo: + keywords["full"] = mo.group(1) + f.close() + except EnvironmentError: + pass + return keywords + + +@register_vcs_handler("git", "keywords") +def git_versions_from_keywords(keywords, tag_prefix, verbose): + """Get version information from git keywords.""" + if not keywords: + raise NotThisMethod("no keywords at all, weird") + refnames = keywords["refnames"].strip() + if refnames.startswith("$Format"): + if verbose: + print("keywords are unexpanded, not using") + raise NotThisMethod("unexpanded keywords, not a git-archive tarball") + refs = set([r.strip() for r in refnames.strip("()").split(",")]) + # starting in git-1.8.3, tags are listed as "tag: foo-1.0" instead of + # just "foo-1.0". If we see a "tag: " prefix, prefer those. + TAG = "tag: " + tags = set([r[len(TAG):] for r in refs if r.startswith(TAG)]) + if not tags: + # Either we're using git < 1.8.3, or there really are no tags. We use + # a heuristic: assume all version tags have a digit. The old git %d + # expansion behaves like git log --decorate=short and strips out the + # refs/heads/ and refs/tags/ prefixes that would let us distinguish + # between branches and tags. By ignoring refnames without digits, we + # filter out many common branch names like "release" and + # "stabilization", as well as "HEAD" and "master". + tags = set([r for r in refs if re.search(r'\d', r)]) + if verbose: + print("discarding '%s', no digits" % ",".join(refs-tags)) + if verbose: + print("likely tags: %s" % ",".join(sorted(tags))) + for ref in sorted(tags): + # sorting will prefer e.g. "2.0" over "2.0rc1" + if ref.startswith(tag_prefix): + r = ref[len(tag_prefix):] + if verbose: + print("picking %s" % r) + return {"version": r, + "full-revisionid": keywords["full"].strip(), + "dirty": False, "error": None + } + # no suitable tags, so version is "0+unknown", but full hex is still there + if verbose: + print("no suitable tags, using unknown + full revision id") + return {"version": "0+unknown", + "full-revisionid": keywords["full"].strip(), + "dirty": False, "error": "no suitable tags"} + + +@register_vcs_handler("git", "pieces_from_vcs") +def git_pieces_from_vcs(tag_prefix, root, verbose, run_command=run_command): + """Get version from 'git describe' in the root of the source tree. + + This only gets called if the git-archive 'subst' keywords were *not* + expanded, and _version.py hasn't already been rewritten with a short + version string, meaning we're inside a checked out source tree. + """ + if not os.path.exists(os.path.join(root, ".git")): + if verbose: + print("no .git in %s" % root) + raise NotThisMethod("no .git directory") + + GITS = ["git"] + if sys.platform == "win32": + GITS = ["git.cmd", "git.exe"] + # if there is a tag matching tag_prefix, this yields TAG-NUM-gHEX[-dirty] + # if there isn't one, this yields HEX[-dirty] (no NUM) + describe_out = run_command(GITS, ["describe", "--tags", "--dirty", + "--always", "--long", + "--match", "%s*" % tag_prefix], + cwd=root) + # --long was added in git-1.5.5 + if describe_out is None: + raise NotThisMethod("'git describe' failed") + describe_out = describe_out.strip() + full_out = run_command(GITS, ["rev-parse", "HEAD"], cwd=root) + if full_out is None: + raise NotThisMethod("'git rev-parse' failed") + full_out = full_out.strip() + + pieces = {} + pieces["long"] = full_out + pieces["short"] = full_out[:7] # maybe improved later + pieces["error"] = None + + # parse describe_out. It will be like TAG-NUM-gHEX[-dirty] or HEX[-dirty] + # TAG might have hyphens. + git_describe = describe_out + + # look for -dirty suffix + dirty = git_describe.endswith("-dirty") + pieces["dirty"] = dirty + if dirty: + git_describe = git_describe[:git_describe.rindex("-dirty")] + + # now we have TAG-NUM-gHEX or HEX + + if "-" in git_describe: + # TAG-NUM-gHEX + mo = re.search(r'^(.+)-(\d+)-g([0-9a-f]+)$', git_describe) + if not mo: + # unparseable. Maybe git-describe is misbehaving? + pieces["error"] = ("unable to parse git-describe output: '%s'" + % describe_out) + return pieces + + # tag + full_tag = mo.group(1) + if not full_tag.startswith(tag_prefix): + if verbose: + fmt = "tag '%s' doesn't start with prefix '%s'" + print(fmt % (full_tag, tag_prefix)) + pieces["error"] = ("tag '%s' doesn't start with prefix '%s'" + % (full_tag, tag_prefix)) + return pieces + pieces["closest-tag"] = full_tag[len(tag_prefix):] + + # distance: number of commits since tag + pieces["distance"] = int(mo.group(2)) + + # commit: short hex revision ID + pieces["short"] = mo.group(3) + + else: + # HEX: no tags + pieces["closest-tag"] = None + count_out = run_command(GITS, ["rev-list", "HEAD", "--count"], + cwd=root) + pieces["distance"] = int(count_out) # total number of commits + + return pieces + + +def plus_or_dot(pieces): + """Return a + if we don't already have one, else return a .""" + if "+" in pieces.get("closest-tag", ""): + return "." + return "+" + + +def render_pep440(pieces): + """Build up version string, with post-release "local version identifier". + + Our goal: TAG[+DISTANCE.gHEX[.dirty]] . Note that if you + get a tagged build and then dirty it, you'll get TAG+0.gHEX.dirty + + Exceptions: + 1: no tags. git_describe was just HEX. 0+untagged.DISTANCE.gHEX[.dirty] + """ + if pieces["closest-tag"]: + rendered = pieces["closest-tag"] + if pieces["distance"] or pieces["dirty"]: + rendered += plus_or_dot(pieces) + rendered += "%d.g%s" % (pieces["distance"], pieces["short"]) + if pieces["dirty"]: + rendered += ".dirty" + else: + # exception #1 + rendered = "0+untagged.%d.g%s" % (pieces["distance"], + pieces["short"]) + if pieces["dirty"]: + rendered += ".dirty" + return rendered + + +def render_pep440_pre(pieces): + """TAG[.post.devDISTANCE] -- No -dirty. + + Exceptions: + 1: no tags. 0.post.devDISTANCE + """ + if pieces["closest-tag"]: + rendered = pieces["closest-tag"] + if pieces["distance"]: + rendered += ".post.dev%d" % pieces["distance"] + else: + # exception #1 + rendered = "0.post.dev%d" % pieces["distance"] + return rendered + + +def render_pep440_post(pieces): + """TAG[.postDISTANCE[.dev0]+gHEX] . + + The ".dev0" means dirty. Note that .dev0 sorts backwards + (a dirty tree will appear "older" than the corresponding clean one), + but you shouldn't be releasing software with -dirty anyways. + + Exceptions: + 1: no tags. 0.postDISTANCE[.dev0] + """ + if pieces["closest-tag"]: + rendered = pieces["closest-tag"] + if pieces["distance"] or pieces["dirty"]: + rendered += ".post%d" % pieces["distance"] + if pieces["dirty"]: + rendered += ".dev0" + rendered += plus_or_dot(pieces) + rendered += "g%s" % pieces["short"] + else: + # exception #1 + rendered = "0.post%d" % pieces["distance"] + if pieces["dirty"]: + rendered += ".dev0" + rendered += "+g%s" % pieces["short"] + return rendered + + +def render_pep440_old(pieces): + """TAG[.postDISTANCE[.dev0]] . + + The ".dev0" means dirty. + + Eexceptions: + 1: no tags. 0.postDISTANCE[.dev0] + """ + if pieces["closest-tag"]: + rendered = pieces["closest-tag"] + if pieces["distance"] or pieces["dirty"]: + rendered += ".post%d" % pieces["distance"] + if pieces["dirty"]: + rendered += ".dev0" + else: + # exception #1 + rendered = "0.post%d" % pieces["distance"] + if pieces["dirty"]: + rendered += ".dev0" + return rendered + + +def render_git_describe(pieces): + """TAG[-DISTANCE-gHEX][-dirty]. + + Like 'git describe --tags --dirty --always'. + + Exceptions: + 1: no tags. HEX[-dirty] (note: no 'g' prefix) + """ + if pieces["closest-tag"]: + rendered = pieces["closest-tag"] + if pieces["distance"]: + rendered += "-%d-g%s" % (pieces["distance"], pieces["short"]) + else: + # exception #1 + rendered = pieces["short"] + if pieces["dirty"]: + rendered += "-dirty" + return rendered + + +def render_git_describe_long(pieces): + """TAG-DISTANCE-gHEX[-dirty]. + + Like 'git describe --tags --dirty --always -long'. + The distance/hash is unconditional. + + Exceptions: + 1: no tags. HEX[-dirty] (note: no 'g' prefix) + """ + if pieces["closest-tag"]: + rendered = pieces["closest-tag"] + rendered += "-%d-g%s" % (pieces["distance"], pieces["short"]) + else: + # exception #1 + rendered = pieces["short"] + if pieces["dirty"]: + rendered += "-dirty" + return rendered + + +def render(pieces, style): + """Render the given version pieces into the requested style.""" + if pieces["error"]: + return {"version": "unknown", + "full-revisionid": pieces.get("long"), + "dirty": None, + "error": pieces["error"]} + + if not style or style == "default": + style = "pep440" # the default + + if style == "pep440": + rendered = render_pep440(pieces) + elif style == "pep440-pre": + rendered = render_pep440_pre(pieces) + elif style == "pep440-post": + rendered = render_pep440_post(pieces) + elif style == "pep440-old": + rendered = render_pep440_old(pieces) + elif style == "git-describe": + rendered = render_git_describe(pieces) + elif style == "git-describe-long": + rendered = render_git_describe_long(pieces) + else: + raise ValueError("unknown style '%s'" % style) + + return {"version": rendered, "full-revisionid": pieces["long"], + "dirty": pieces["dirty"], "error": None} + + +def get_versions(): + """Get version information or return default if unable to do so.""" + # I am in _version.py, which lives at ROOT/VERSIONFILE_SOURCE. If we have + # __file__, we can work backwards from there to the root. Some + # py2exe/bbfreeze/non-CPython implementations don't do __file__, in which + # case we can only use expanded keywords. + + cfg = get_config() + verbose = cfg.verbose + + try: + return git_versions_from_keywords(get_keywords(), cfg.tag_prefix, + verbose) + except NotThisMethod: + pass + + try: + root = os.path.realpath(__file__) + # versionfile_source is the relative path from the top of the source + # tree (where the .git directory might live) to this file. Invert + # this to find the root from __file__. + for i in cfg.versionfile_source.split('/'): + root = os.path.dirname(root) + except NameError: + return {"version": "0+unknown", "full-revisionid": None, + "dirty": None, + "error": "unable to find root of source tree"} + + try: + pieces = git_pieces_from_vcs(cfg.tag_prefix, root, verbose) + return render(pieces, cfg.style) + except NotThisMethod: + pass + + try: + if cfg.parentdir_prefix: + return versions_from_parentdir(cfg.parentdir_prefix, root, verbose) + except NotThisMethod: + pass + + return {"version": "0+unknown", "full-revisionid": None, + "dirty": None, + "error": "unable to compute version"} diff --git a/setup.cfg b/setup.cfg index 2a9acf1..96b08c0 100644 --- a/setup.cfg +++ b/setup.cfg @@ -1,2 +1,10 @@ [bdist_wheel] universal = 1 + +[versioneer] +VCS = git +style = pep440 +versionfile_source = geopandas/_version.py +versionfile_build = geopandas/_version.py +tag_prefix = v +parentdir_prefix = geopandas- diff --git a/setup.py b/setup.py index eed0c17..1b63619 100644 --- a/setup.py +++ b/setup.py @@ -13,6 +13,8 @@ try: except ImportError: from distutils.core import setup +import versioneer + LONG_DESCRIPTION = """GeoPandas is a project to add support for geographic data to `pandas`_ objects. @@ -27,61 +29,11 @@ such as PostGIS. .. _shapely: http://toblerity.github.io/shapely """ -MAJOR = 0 -MINOR = 1 -MICRO = 0 -ISRELEASED = False -VERSION = '%d.%d.%d' % (MAJOR, MINOR, MICRO) -QUALIFIER = '' if os.environ.get('READTHEDOCS', False) == 'True': INSTALL_REQUIRES = [] else: INSTALL_REQUIRES = ['pandas', 'shapely', 'fiona', 'descartes', 'pyproj'] -FULLVERSION = VERSION -if not ISRELEASED: - FULLVERSION += '.dev' - try: - import subprocess - try: - pipe = subprocess.Popen(["git", "rev-parse", "--short", "HEAD"], - stdout=subprocess.PIPE).stdout - except OSError: - # msysgit compatibility - pipe = subprocess.Popen( - ["git.cmd", "describe", "HEAD"], - stdout=subprocess.PIPE).stdout - rev = pipe.read().strip() - # makes distutils blow up on Python 2.7 - if sys.version_info[0] >= 3: - rev = rev.decode('ascii') - - FULLVERSION = '%d.%d.%d.dev-%s' % (MAJOR, MINOR, MICRO, rev) - - except: - warnings.warn("WARNING: Couldn't get git revision") -else: - FULLVERSION += QUALIFIER - - -def write_version_py(filename=None): - cnt = """\ -version = '%s' -short_version = '%s' -""" - if not filename: - filename = os.path.join( - os.path.dirname(__file__), 'geopandas', 'version.py') - - a = open(filename, 'w') - try: - a.write(cnt % (FULLVERSION, VERSION)) - finally: - a.close() - -write_version_py() - - # get all data dirs in the datasets module data_files = [] @@ -91,7 +43,7 @@ for item in os.listdir("geopandas/datasets"): data_files.append(os.path.join("datasets", item, '*')) setup(name='geopandas', - version=FULLVERSION, + version=versioneer.get_version(), description='Geographic pandas extensions', license='BSD', author='Kelsey Jordahl', diff --git a/versioneer.py b/versioneer.py new file mode 100644 index 0000000..7ed2a21 --- /dev/null +++ b/versioneer.py @@ -0,0 +1,1774 @@ + +# Version: 0.16 + +"""The Versioneer - like a rocketeer, but for versions. + +The Versioneer +============== + +* like a rocketeer, but for versions! +* https://github.com/warner/python-versioneer +* Brian Warner +* License: Public Domain +* Compatible With: python2.6, 2.7, 3.3, 3.4, 3.5, and pypy +* [![Latest Version] +(https://pypip.in/version/versioneer/badge.svg?style=flat) +](https://pypi.python.org/pypi/versioneer/) +* [![Build Status] +(https://travis-ci.org/warner/python-versioneer.png?branch=master) +](https://travis-ci.org/warner/python-versioneer) + +This is a tool for managing a recorded version number in distutils-based +python projects. The goal is to remove the tedious and error-prone "update +the embedded version string" step from your release process. Making a new +release should be as easy as recording a new tag in your version-control +system, and maybe making new tarballs. + + +## Quick Install + +* `pip install versioneer` to somewhere to your $PATH +* add a `[versioneer]` section to your setup.cfg (see below) +* run `versioneer install` in your source tree, commit the results + +## Version Identifiers + +Source trees come from a variety of places: + +* a version-control system checkout (mostly used by developers) +* a nightly tarball, produced by build automation +* a snapshot tarball, produced by a web-based VCS browser, like github's + "tarball from tag" feature +* a release tarball, produced by "setup.py sdist", distributed through PyPI + +Within each source tree, the version identifier (either a string or a number, +this tool is format-agnostic) can come from a variety of places: + +* ask the VCS tool itself, e.g. "git describe" (for checkouts), which knows + about recent "tags" and an absolute revision-id +* the name of the directory into which the tarball was unpacked +* an expanded VCS keyword ($Id$, etc) +* a `_version.py` created by some earlier build step + +For released software, the version identifier is closely related to a VCS +tag. Some projects use tag names that include more than just the version +string (e.g. "myproject-1.2" instead of just "1.2"), in which case the tool +needs to strip the tag prefix to extract the version identifier. For +unreleased software (between tags), the version identifier should provide +enough information to help developers recreate the same tree, while also +giving them an idea of roughly how old the tree is (after version 1.2, before +version 1.3). Many VCS systems can report a description that captures this, +for example `git describe --tags --dirty --always` reports things like +"0.7-1-g574ab98-dirty" to indicate that the checkout is one revision past the +0.7 tag, has a unique revision id of "574ab98", and is "dirty" (it has +uncommitted changes. + +The version identifier is used for multiple purposes: + +* to allow the module to self-identify its version: `myproject.__version__` +* to choose a name and prefix for a 'setup.py sdist' tarball + +## Theory of Operation + +Versioneer works by adding a special `_version.py` file into your source +tree, where your `__init__.py` can import it. This `_version.py` knows how to +dynamically ask the VCS tool for version information at import time. + +`_version.py` also contains `$Revision$` markers, and the installation +process marks `_version.py` to have this marker rewritten with a tag name +during the `git archive` command. As a result, generated tarballs will +contain enough information to get the proper version. + +To allow `setup.py` to compute a version too, a `versioneer.py` is added to +the top level of your source tree, next to `setup.py` and the `setup.cfg` +that configures it. This overrides several distutils/setuptools commands to +compute the version when invoked, and changes `setup.py build` and `setup.py +sdist` to replace `_version.py` with a small static file that contains just +the generated version data. + +## Installation + +First, decide on values for the following configuration variables: + +* `VCS`: the version control system you use. Currently accepts "git". + +* `style`: the style of version string to be produced. See "Styles" below for + details. Defaults to "pep440", which looks like + `TAG[+DISTANCE.gSHORTHASH[.dirty]]`. + +* `versionfile_source`: + + A project-relative pathname into which the generated version strings should + be written. This is usually a `_version.py` next to your project's main + `__init__.py` file, so it can be imported at runtime. If your project uses + `src/myproject/__init__.py`, this should be `src/myproject/_version.py`. + This file should be checked in to your VCS as usual: the copy created below + by `setup.py setup_versioneer` will include code that parses expanded VCS + keywords in generated tarballs. The 'build' and 'sdist' commands will + replace it with a copy that has just the calculated version string. + + This must be set even if your project does not have any modules (and will + therefore never import `_version.py`), since "setup.py sdist" -based trees + still need somewhere to record the pre-calculated version strings. Anywhere + in the source tree should do. If there is a `__init__.py` next to your + `_version.py`, the `setup.py setup_versioneer` command (described below) + will append some `__version__`-setting assignments, if they aren't already + present. + +* `versionfile_build`: + + Like `versionfile_source`, but relative to the build directory instead of + the source directory. These will differ when your setup.py uses + 'package_dir='. If you have `package_dir={'myproject': 'src/myproject'}`, + then you will probably have `versionfile_build='myproject/_version.py'` and + `versionfile_source='src/myproject/_version.py'`. + + If this is set to None, then `setup.py build` will not attempt to rewrite + any `_version.py` in the built tree. If your project does not have any + libraries (e.g. if it only builds a script), then you should use + `versionfile_build = None`. To actually use the computed version string, + your `setup.py` will need to override `distutils.command.build_scripts` + with a subclass that explicitly inserts a copy of + `versioneer.get_version()` into your script file. See + `test/demoapp-script-only/setup.py` for an example. + +* `tag_prefix`: + + a string, like 'PROJECTNAME-', which appears at the start of all VCS tags. + If your tags look like 'myproject-1.2.0', then you should use + tag_prefix='myproject-'. If you use unprefixed tags like '1.2.0', this + should be an empty string, using either `tag_prefix=` or `tag_prefix=''`. + +* `parentdir_prefix`: + + a optional string, frequently the same as tag_prefix, which appears at the + start of all unpacked tarball filenames. If your tarball unpacks into + 'myproject-1.2.0', this should be 'myproject-'. To disable this feature, + just omit the field from your `setup.cfg`. + +This tool provides one script, named `versioneer`. That script has one mode, +"install", which writes a copy of `versioneer.py` into the current directory +and runs `versioneer.py setup` to finish the installation. + +To versioneer-enable your project: + +* 1: Modify your `setup.cfg`, adding a section named `[versioneer]` and + populating it with the configuration values you decided earlier (note that + the option names are not case-sensitive): + + ```` + [versioneer] + VCS = git + style = pep440 + versionfile_source = src/myproject/_version.py + versionfile_build = myproject/_version.py + tag_prefix = + parentdir_prefix = myproject- + ```` + +* 2: Run `versioneer install`. This will do the following: + + * copy `versioneer.py` into the top of your source tree + * create `_version.py` in the right place (`versionfile_source`) + * modify your `__init__.py` (if one exists next to `_version.py`) to define + `__version__` (by calling a function from `_version.py`) + * modify your `MANIFEST.in` to include both `versioneer.py` and the + generated `_version.py` in sdist tarballs + + `versioneer install` will complain about any problems it finds with your + `setup.py` or `setup.cfg`. Run it multiple times until you have fixed all + the problems. + +* 3: add a `import versioneer` to your setup.py, and add the following + arguments to the setup() call: + + version=versioneer.get_version(), + cmdclass=versioneer.get_cmdclass(), + +* 4: commit these changes to your VCS. To make sure you won't forget, + `versioneer install` will mark everything it touched for addition using + `git add`. Don't forget to add `setup.py` and `setup.cfg` too. + +## Post-Installation Usage + +Once established, all uses of your tree from a VCS checkout should get the +current version string. All generated tarballs should include an embedded +version string (so users who unpack them will not need a VCS tool installed). + +If you distribute your project through PyPI, then the release process should +boil down to two steps: + +* 1: git tag 1.0 +* 2: python setup.py register sdist upload + +If you distribute it through github (i.e. users use github to generate +tarballs with `git archive`), the process is: + +* 1: git tag 1.0 +* 2: git push; git push --tags + +Versioneer will report "0+untagged.NUMCOMMITS.gHASH" until your tree has at +least one tag in its history. + +## Version-String Flavors + +Code which uses Versioneer can learn about its version string at runtime by +importing `_version` from your main `__init__.py` file and running the +`get_versions()` function. From the "outside" (e.g. in `setup.py`), you can +import the top-level `versioneer.py` and run `get_versions()`. + +Both functions return a dictionary with different flavors of version +information: + +* `['version']`: A condensed version string, rendered using the selected + style. This is the most commonly used value for the project's version + string. The default "pep440" style yields strings like `0.11`, + `0.11+2.g1076c97`, or `0.11+2.g1076c97.dirty`. See the "Styles" section + below for alternative styles. + +* `['full-revisionid']`: detailed revision identifier. For Git, this is the + full SHA1 commit id, e.g. "1076c978a8d3cfc70f408fe5974aa6c092c949ac". + +* `['dirty']`: a boolean, True if the tree has uncommitted changes. Note that + this is only accurate if run in a VCS checkout, otherwise it is likely to + be False or None + +* `['error']`: if the version string could not be computed, this will be set + to a string describing the problem, otherwise it will be None. It may be + useful to throw an exception in setup.py if this is set, to avoid e.g. + creating tarballs with a version string of "unknown". + +Some variants are more useful than others. Including `full-revisionid` in a +bug report should allow developers to reconstruct the exact code being tested +(or indicate the presence of local changes that should be shared with the +developers). `version` is suitable for display in an "about" box or a CLI +`--version` output: it can be easily compared against release notes and lists +of bugs fixed in various releases. + +The installer adds the following text to your `__init__.py` to place a basic +version in `YOURPROJECT.__version__`: + + from ._version import get_versions + __version__ = get_versions()['version'] + del get_versions + +## Styles + +The setup.cfg `style=` configuration controls how the VCS information is +rendered into a version string. + +The default style, "pep440", produces a PEP440-compliant string, equal to the +un-prefixed tag name for actual releases, and containing an additional "local +version" section with more detail for in-between builds. For Git, this is +TAG[+DISTANCE.gHEX[.dirty]] , using information from `git describe --tags +--dirty --always`. For example "0.11+2.g1076c97.dirty" indicates that the +tree is like the "1076c97" commit but has uncommitted changes (".dirty"), and +that this commit is two revisions ("+2") beyond the "0.11" tag. For released +software (exactly equal to a known tag), the identifier will only contain the +stripped tag, e.g. "0.11". + +Other styles are available. See details.md in the Versioneer source tree for +descriptions. + +## Debugging + +Versioneer tries to avoid fatal errors: if something goes wrong, it will tend +to return a version of "0+unknown". To investigate the problem, run `setup.py +version`, which will run the version-lookup code in a verbose mode, and will +display the full contents of `get_versions()` (including the `error` string, +which may help identify what went wrong). + +## Updating Versioneer + +To upgrade your project to a new release of Versioneer, do the following: + +* install the new Versioneer (`pip install -U versioneer` or equivalent) +* edit `setup.cfg`, if necessary, to include any new configuration settings + indicated by the release notes +* re-run `versioneer install` in your source tree, to replace + `SRC/_version.py` +* commit any changed files + +### Upgrading to 0.16 + +Nothing special. + +### Upgrading to 0.15 + +Starting with this version, Versioneer is configured with a `[versioneer]` +section in your `setup.cfg` file. Earlier versions required the `setup.py` to +set attributes on the `versioneer` module immediately after import. The new +version will refuse to run (raising an exception during import) until you +have provided the necessary `setup.cfg` section. + +In addition, the Versioneer package provides an executable named +`versioneer`, and the installation process is driven by running `versioneer +install`. In 0.14 and earlier, the executable was named +`versioneer-installer` and was run without an argument. + +### Upgrading to 0.14 + +0.14 changes the format of the version string. 0.13 and earlier used +hyphen-separated strings like "0.11-2-g1076c97-dirty". 0.14 and beyond use a +plus-separated "local version" section strings, with dot-separated +components, like "0.11+2.g1076c97". PEP440-strict tools did not like the old +format, but should be ok with the new one. + +### Upgrading from 0.11 to 0.12 + +Nothing special. + +### Upgrading from 0.10 to 0.11 + +You must add a `versioneer.VCS = "git"` to your `setup.py` before re-running +`setup.py setup_versioneer`. This will enable the use of additional +version-control systems (SVN, etc) in the future. + +## Future Directions + +This tool is designed to make it easily extended to other version-control +systems: all VCS-specific components are in separate directories like +src/git/ . The top-level `versioneer.py` script is assembled from these +components by running make-versioneer.py . In the future, make-versioneer.py +will take a VCS name as an argument, and will construct a version of +`versioneer.py` that is specific to the given VCS. It might also take the +configuration arguments that are currently provided manually during +installation by editing setup.py . Alternatively, it might go the other +direction and include code from all supported VCS systems, reducing the +number of intermediate scripts. + + +## License + +To make Versioneer easier to embed, all its code is dedicated to the public +domain. The `_version.py` that it creates is also in the public domain. +Specifically, both are released under the Creative Commons "Public Domain +Dedication" license (CC0-1.0), as described in +https://creativecommons.org/publicdomain/zero/1.0/ . + +""" + +from __future__ import print_function +try: + import configparser +except ImportError: + import ConfigParser as configparser +import errno +import json +import os +import re +import subprocess +import sys + + +class VersioneerConfig: + """Container for Versioneer configuration parameters.""" + + +def get_root(): + """Get the project root directory. + + We require that all commands are run from the project root, i.e. the + directory that contains setup.py, setup.cfg, and versioneer.py . + """ + root = os.path.realpath(os.path.abspath(os.getcwd())) + setup_py = os.path.join(root, "setup.py") + versioneer_py = os.path.join(root, "versioneer.py") + if not (os.path.exists(setup_py) or os.path.exists(versioneer_py)): + # allow 'python path/to/setup.py COMMAND' + root = os.path.dirname(os.path.realpath(os.path.abspath(sys.argv[0]))) + setup_py = os.path.join(root, "setup.py") + versioneer_py = os.path.join(root, "versioneer.py") + if not (os.path.exists(setup_py) or os.path.exists(versioneer_py)): + err = ("Versioneer was unable to run the project root directory. " + "Versioneer requires setup.py to be executed from " + "its immediate directory (like 'python setup.py COMMAND'), " + "or in a way that lets it use sys.argv[0] to find the root " + "(like 'python path/to/setup.py COMMAND').") + raise VersioneerBadRootError(err) + try: + # Certain runtime workflows (setup.py install/develop in a setuptools + # tree) execute all dependencies in a single python process, so + # "versioneer" may be imported multiple times, and python's shared + # module-import table will cache the first one. So we can't use + # os.path.dirname(__file__), as that will find whichever + # versioneer.py was first imported, even in later projects. + me = os.path.realpath(os.path.abspath(__file__)) + if os.path.splitext(me)[0] != os.path.splitext(versioneer_py)[0]: + print("Warning: build in %s is using versioneer.py from %s" + % (os.path.dirname(me), versioneer_py)) + except NameError: + pass + return root + + +def get_config_from_root(root): + """Read the project setup.cfg file to determine Versioneer config.""" + # This might raise EnvironmentError (if setup.cfg is missing), or + # configparser.NoSectionError (if it lacks a [versioneer] section), or + # configparser.NoOptionError (if it lacks "VCS="). See the docstring at + # the top of versioneer.py for instructions on writing your setup.cfg . + setup_cfg = os.path.join(root, "setup.cfg") + parser = configparser.SafeConfigParser() + with open(setup_cfg, "r") as f: + parser.readfp(f) + VCS = parser.get("versioneer", "VCS") # mandatory + + def get(parser, name): + if parser.has_option("versioneer", name): + return parser.get("versioneer", name) + return None + cfg = VersioneerConfig() + cfg.VCS = VCS + cfg.style = get(parser, "style") or "" + cfg.versionfile_source = get(parser, "versionfile_source") + cfg.versionfile_build = get(parser, "versionfile_build") + cfg.tag_prefix = get(parser, "tag_prefix") + if cfg.tag_prefix in ("''", '""'): + cfg.tag_prefix = "" + cfg.parentdir_prefix = get(parser, "parentdir_prefix") + cfg.verbose = get(parser, "verbose") + return cfg + + +class NotThisMethod(Exception): + """Exception raised if a method is not valid for the current scenario.""" + +# these dictionaries contain VCS-specific tools +LONG_VERSION_PY = {} +HANDLERS = {} + + +def register_vcs_handler(vcs, method): # decorator + """Decorator to mark a method as the handler for a particular VCS.""" + def decorate(f): + """Store f in HANDLERS[vcs][method].""" + if vcs not in HANDLERS: + HANDLERS[vcs] = {} + HANDLERS[vcs][method] = f + return f + return decorate + + +def run_command(commands, args, cwd=None, verbose=False, hide_stderr=False): + """Call the given command(s).""" + assert isinstance(commands, list) + p = None + for c in commands: + try: + dispcmd = str([c] + args) + # remember shell=False, so use git.cmd on windows, not just git + p = subprocess.Popen([c] + args, cwd=cwd, stdout=subprocess.PIPE, + stderr=(subprocess.PIPE if hide_stderr + else None)) + break + except EnvironmentError: + e = sys.exc_info()[1] + if e.errno == errno.ENOENT: + continue + if verbose: + print("unable to run %s" % dispcmd) + print(e) + return None + else: + if verbose: + print("unable to find command, tried %s" % (commands,)) + return None + stdout = p.communicate()[0].strip() + if sys.version_info[0] >= 3: + stdout = stdout.decode() + if p.returncode != 0: + if verbose: + print("unable to run %s (error)" % dispcmd) + return None + return stdout +LONG_VERSION_PY['git'] = ''' +# This file helps to compute a version number in source trees obtained from +# git-archive tarball (such as those provided by githubs download-from-tag +# feature). Distribution tarballs (built by setup.py sdist) and build +# directories (produced by setup.py build) will contain a much shorter file +# that just contains the computed version number. + +# This file is released into the public domain. Generated by +# versioneer-0.16 (https://github.com/warner/python-versioneer) + +"""Git implementation of _version.py.""" + +import errno +import os +import re +import subprocess +import sys + + +def get_keywords(): + """Get the keywords needed to look up the version information.""" + # these strings will be replaced by git during git-archive. + # setup.py/versioneer.py will grep for the variable names, so they must + # each be defined on a line of their own. _version.py will just call + # get_keywords(). + git_refnames = "%(DOLLAR)sFormat:%%d%(DOLLAR)s" + git_full = "%(DOLLAR)sFormat:%%H%(DOLLAR)s" + keywords = {"refnames": git_refnames, "full": git_full} + return keywords + + +class VersioneerConfig: + """Container for Versioneer configuration parameters.""" + + +def get_config(): + """Create, populate and return the VersioneerConfig() object.""" + # these strings are filled in when 'setup.py versioneer' creates + # _version.py + cfg = VersioneerConfig() + cfg.VCS = "git" + cfg.style = "%(STYLE)s" + cfg.tag_prefix = "%(TAG_PREFIX)s" + cfg.parentdir_prefix = "%(PARENTDIR_PREFIX)s" + cfg.versionfile_source = "%(VERSIONFILE_SOURCE)s" + cfg.verbose = False + return cfg + + +class NotThisMethod(Exception): + """Exception raised if a method is not valid for the current scenario.""" + + +LONG_VERSION_PY = {} +HANDLERS = {} + + +def register_vcs_handler(vcs, method): # decorator + """Decorator to mark a method as the handler for a particular VCS.""" + def decorate(f): + """Store f in HANDLERS[vcs][method].""" + if vcs not in HANDLERS: + HANDLERS[vcs] = {} + HANDLERS[vcs][method] = f + return f + return decorate + + +def run_command(commands, args, cwd=None, verbose=False, hide_stderr=False): + """Call the given command(s).""" + assert isinstance(commands, list) + p = None + for c in commands: + try: + dispcmd = str([c] + args) + # remember shell=False, so use git.cmd on windows, not just git + p = subprocess.Popen([c] + args, cwd=cwd, stdout=subprocess.PIPE, + stderr=(subprocess.PIPE if hide_stderr + else None)) + break + except EnvironmentError: + e = sys.exc_info()[1] + if e.errno == errno.ENOENT: + continue + if verbose: + print("unable to run %%s" %% dispcmd) + print(e) + return None + else: + if verbose: + print("unable to find command, tried %%s" %% (commands,)) + return None + stdout = p.communicate()[0].strip() + if sys.version_info[0] >= 3: + stdout = stdout.decode() + if p.returncode != 0: + if verbose: + print("unable to run %%s (error)" %% dispcmd) + return None + return stdout + + +def versions_from_parentdir(parentdir_prefix, root, verbose): + """Try to determine the version from the parent directory name. + + Source tarballs conventionally unpack into a directory that includes + both the project name and a version string. + """ + dirname = os.path.basename(root) + if not dirname.startswith(parentdir_prefix): + if verbose: + print("guessing rootdir is '%%s', but '%%s' doesn't start with " + "prefix '%%s'" %% (root, dirname, parentdir_prefix)) + raise NotThisMethod("rootdir doesn't start with parentdir_prefix") + return {"version": dirname[len(parentdir_prefix):], + "full-revisionid": None, + "dirty": False, "error": None} + + +@register_vcs_handler("git", "get_keywords") +def git_get_keywords(versionfile_abs): + """Extract version information from the given file.""" + # the code embedded in _version.py can just fetch the value of these + # keywords. When used from setup.py, we don't want to import _version.py, + # so we do it with a regexp instead. This function is not used from + # _version.py. + keywords = {} + try: + f = open(versionfile_abs, "r") + for line in f.readlines(): + if line.strip().startswith("git_refnames ="): + mo = re.search(r'=\s*"(.*)"', line) + if mo: + keywords["refnames"] = mo.group(1) + if line.strip().startswith("git_full ="): + mo = re.search(r'=\s*"(.*)"', line) + if mo: + keywords["full"] = mo.group(1) + f.close() + except EnvironmentError: + pass + return keywords + + +@register_vcs_handler("git", "keywords") +def git_versions_from_keywords(keywords, tag_prefix, verbose): + """Get version information from git keywords.""" + if not keywords: + raise NotThisMethod("no keywords at all, weird") + refnames = keywords["refnames"].strip() + if refnames.startswith("$Format"): + if verbose: + print("keywords are unexpanded, not using") + raise NotThisMethod("unexpanded keywords, not a git-archive tarball") + refs = set([r.strip() for r in refnames.strip("()").split(",")]) + # starting in git-1.8.3, tags are listed as "tag: foo-1.0" instead of + # just "foo-1.0". If we see a "tag: " prefix, prefer those. + TAG = "tag: " + tags = set([r[len(TAG):] for r in refs if r.startswith(TAG)]) + if not tags: + # Either we're using git < 1.8.3, or there really are no tags. We use + # a heuristic: assume all version tags have a digit. The old git %%d + # expansion behaves like git log --decorate=short and strips out the + # refs/heads/ and refs/tags/ prefixes that would let us distinguish + # between branches and tags. By ignoring refnames without digits, we + # filter out many common branch names like "release" and + # "stabilization", as well as "HEAD" and "master". + tags = set([r for r in refs if re.search(r'\d', r)]) + if verbose: + print("discarding '%%s', no digits" %% ",".join(refs-tags)) + if verbose: + print("likely tags: %%s" %% ",".join(sorted(tags))) + for ref in sorted(tags): + # sorting will prefer e.g. "2.0" over "2.0rc1" + if ref.startswith(tag_prefix): + r = ref[len(tag_prefix):] + if verbose: + print("picking %%s" %% r) + return {"version": r, + "full-revisionid": keywords["full"].strip(), + "dirty": False, "error": None + } + # no suitable tags, so version is "0+unknown", but full hex is still there + if verbose: + print("no suitable tags, using unknown + full revision id") + return {"version": "0+unknown", + "full-revisionid": keywords["full"].strip(), + "dirty": False, "error": "no suitable tags"} + + +@register_vcs_handler("git", "pieces_from_vcs") +def git_pieces_from_vcs(tag_prefix, root, verbose, run_command=run_command): + """Get version from 'git describe' in the root of the source tree. + + This only gets called if the git-archive 'subst' keywords were *not* + expanded, and _version.py hasn't already been rewritten with a short + version string, meaning we're inside a checked out source tree. + """ + if not os.path.exists(os.path.join(root, ".git")): + if verbose: + print("no .git in %%s" %% root) + raise NotThisMethod("no .git directory") + + GITS = ["git"] + if sys.platform == "win32": + GITS = ["git.cmd", "git.exe"] + # if there is a tag matching tag_prefix, this yields TAG-NUM-gHEX[-dirty] + # if there isn't one, this yields HEX[-dirty] (no NUM) + describe_out = run_command(GITS, ["describe", "--tags", "--dirty", + "--always", "--long", + "--match", "%%s*" %% tag_prefix], + cwd=root) + # --long was added in git-1.5.5 + if describe_out is None: + raise NotThisMethod("'git describe' failed") + describe_out = describe_out.strip() + full_out = run_command(GITS, ["rev-parse", "HEAD"], cwd=root) + if full_out is None: + raise NotThisMethod("'git rev-parse' failed") + full_out = full_out.strip() + + pieces = {} + pieces["long"] = full_out + pieces["short"] = full_out[:7] # maybe improved later + pieces["error"] = None + + # parse describe_out. It will be like TAG-NUM-gHEX[-dirty] or HEX[-dirty] + # TAG might have hyphens. + git_describe = describe_out + + # look for -dirty suffix + dirty = git_describe.endswith("-dirty") + pieces["dirty"] = dirty + if dirty: + git_describe = git_describe[:git_describe.rindex("-dirty")] + + # now we have TAG-NUM-gHEX or HEX + + if "-" in git_describe: + # TAG-NUM-gHEX + mo = re.search(r'^(.+)-(\d+)-g([0-9a-f]+)$', git_describe) + if not mo: + # unparseable. Maybe git-describe is misbehaving? + pieces["error"] = ("unable to parse git-describe output: '%%s'" + %% describe_out) + return pieces + + # tag + full_tag = mo.group(1) + if not full_tag.startswith(tag_prefix): + if verbose: + fmt = "tag '%%s' doesn't start with prefix '%%s'" + print(fmt %% (full_tag, tag_prefix)) + pieces["error"] = ("tag '%%s' doesn't start with prefix '%%s'" + %% (full_tag, tag_prefix)) + return pieces + pieces["closest-tag"] = full_tag[len(tag_prefix):] + + # distance: number of commits since tag + pieces["distance"] = int(mo.group(2)) + + # commit: short hex revision ID + pieces["short"] = mo.group(3) + + else: + # HEX: no tags + pieces["closest-tag"] = None + count_out = run_command(GITS, ["rev-list", "HEAD", "--count"], + cwd=root) + pieces["distance"] = int(count_out) # total number of commits + + return pieces + + +def plus_or_dot(pieces): + """Return a + if we don't already have one, else return a .""" + if "+" in pieces.get("closest-tag", ""): + return "." + return "+" + + +def render_pep440(pieces): + """Build up version string, with post-release "local version identifier". + + Our goal: TAG[+DISTANCE.gHEX[.dirty]] . Note that if you + get a tagged build and then dirty it, you'll get TAG+0.gHEX.dirty + + Exceptions: + 1: no tags. git_describe was just HEX. 0+untagged.DISTANCE.gHEX[.dirty] + """ + if pieces["closest-tag"]: + rendered = pieces["closest-tag"] + if pieces["distance"] or pieces["dirty"]: + rendered += plus_or_dot(pieces) + rendered += "%%d.g%%s" %% (pieces["distance"], pieces["short"]) + if pieces["dirty"]: + rendered += ".dirty" + else: + # exception #1 + rendered = "0+untagged.%%d.g%%s" %% (pieces["distance"], + pieces["short"]) + if pieces["dirty"]: + rendered += ".dirty" + return rendered + + +def render_pep440_pre(pieces): + """TAG[.post.devDISTANCE] -- No -dirty. + + Exceptions: + 1: no tags. 0.post.devDISTANCE + """ + if pieces["closest-tag"]: + rendered = pieces["closest-tag"] + if pieces["distance"]: + rendered += ".post.dev%%d" %% pieces["distance"] + else: + # exception #1 + rendered = "0.post.dev%%d" %% pieces["distance"] + return rendered + + +def render_pep440_post(pieces): + """TAG[.postDISTANCE[.dev0]+gHEX] . + + The ".dev0" means dirty. Note that .dev0 sorts backwards + (a dirty tree will appear "older" than the corresponding clean one), + but you shouldn't be releasing software with -dirty anyways. + + Exceptions: + 1: no tags. 0.postDISTANCE[.dev0] + """ + if pieces["closest-tag"]: + rendered = pieces["closest-tag"] + if pieces["distance"] or pieces["dirty"]: + rendered += ".post%%d" %% pieces["distance"] + if pieces["dirty"]: + rendered += ".dev0" + rendered += plus_or_dot(pieces) + rendered += "g%%s" %% pieces["short"] + else: + # exception #1 + rendered = "0.post%%d" %% pieces["distance"] + if pieces["dirty"]: + rendered += ".dev0" + rendered += "+g%%s" %% pieces["short"] + return rendered + + +def render_pep440_old(pieces): + """TAG[.postDISTANCE[.dev0]] . + + The ".dev0" means dirty. + + Eexceptions: + 1: no tags. 0.postDISTANCE[.dev0] + """ + if pieces["closest-tag"]: + rendered = pieces["closest-tag"] + if pieces["distance"] or pieces["dirty"]: + rendered += ".post%%d" %% pieces["distance"] + if pieces["dirty"]: + rendered += ".dev0" + else: + # exception #1 + rendered = "0.post%%d" %% pieces["distance"] + if pieces["dirty"]: + rendered += ".dev0" + return rendered + + +def render_git_describe(pieces): + """TAG[-DISTANCE-gHEX][-dirty]. + + Like 'git describe --tags --dirty --always'. + + Exceptions: + 1: no tags. HEX[-dirty] (note: no 'g' prefix) + """ + if pieces["closest-tag"]: + rendered = pieces["closest-tag"] + if pieces["distance"]: + rendered += "-%%d-g%%s" %% (pieces["distance"], pieces["short"]) + else: + # exception #1 + rendered = pieces["short"] + if pieces["dirty"]: + rendered += "-dirty" + return rendered + + +def render_git_describe_long(pieces): + """TAG-DISTANCE-gHEX[-dirty]. + + Like 'git describe --tags --dirty --always -long'. + The distance/hash is unconditional. + + Exceptions: + 1: no tags. HEX[-dirty] (note: no 'g' prefix) + """ + if pieces["closest-tag"]: + rendered = pieces["closest-tag"] + rendered += "-%%d-g%%s" %% (pieces["distance"], pieces["short"]) + else: + # exception #1 + rendered = pieces["short"] + if pieces["dirty"]: + rendered += "-dirty" + return rendered + + +def render(pieces, style): + """Render the given version pieces into the requested style.""" + if pieces["error"]: + return {"version": "unknown", + "full-revisionid": pieces.get("long"), + "dirty": None, + "error": pieces["error"]} + + if not style or style == "default": + style = "pep440" # the default + + if style == "pep440": + rendered = render_pep440(pieces) + elif style == "pep440-pre": + rendered = render_pep440_pre(pieces) + elif style == "pep440-post": + rendered = render_pep440_post(pieces) + elif style == "pep440-old": + rendered = render_pep440_old(pieces) + elif style == "git-describe": + rendered = render_git_describe(pieces) + elif style == "git-describe-long": + rendered = render_git_describe_long(pieces) + else: + raise ValueError("unknown style '%%s'" %% style) + + return {"version": rendered, "full-revisionid": pieces["long"], + "dirty": pieces["dirty"], "error": None} + + +def get_versions(): + """Get version information or return default if unable to do so.""" + # I am in _version.py, which lives at ROOT/VERSIONFILE_SOURCE. If we have + # __file__, we can work backwards from there to the root. Some + # py2exe/bbfreeze/non-CPython implementations don't do __file__, in which + # case we can only use expanded keywords. + + cfg = get_config() + verbose = cfg.verbose + + try: + return git_versions_from_keywords(get_keywords(), cfg.tag_prefix, + verbose) + except NotThisMethod: + pass + + try: + root = os.path.realpath(__file__) + # versionfile_source is the relative path from the top of the source + # tree (where the .git directory might live) to this file. Invert + # this to find the root from __file__. + for i in cfg.versionfile_source.split('/'): + root = os.path.dirname(root) + except NameError: + return {"version": "0+unknown", "full-revisionid": None, + "dirty": None, + "error": "unable to find root of source tree"} + + try: + pieces = git_pieces_from_vcs(cfg.tag_prefix, root, verbose) + return render(pieces, cfg.style) + except NotThisMethod: + pass + + try: + if cfg.parentdir_prefix: + return versions_from_parentdir(cfg.parentdir_prefix, root, verbose) + except NotThisMethod: + pass + + return {"version": "0+unknown", "full-revisionid": None, + "dirty": None, + "error": "unable to compute version"} +''' + + +@register_vcs_handler("git", "get_keywords") +def git_get_keywords(versionfile_abs): + """Extract version information from the given file.""" + # the code embedded in _version.py can just fetch the value of these + # keywords. When used from setup.py, we don't want to import _version.py, + # so we do it with a regexp instead. This function is not used from + # _version.py. + keywords = {} + try: + f = open(versionfile_abs, "r") + for line in f.readlines(): + if line.strip().startswith("git_refnames ="): + mo = re.search(r'=\s*"(.*)"', line) + if mo: + keywords["refnames"] = mo.group(1) + if line.strip().startswith("git_full ="): + mo = re.search(r'=\s*"(.*)"', line) + if mo: + keywords["full"] = mo.group(1) + f.close() + except EnvironmentError: + pass + return keywords + + +@register_vcs_handler("git", "keywords") +def git_versions_from_keywords(keywords, tag_prefix, verbose): + """Get version information from git keywords.""" + if not keywords: + raise NotThisMethod("no keywords at all, weird") + refnames = keywords["refnames"].strip() + if refnames.startswith("$Format"): + if verbose: + print("keywords are unexpanded, not using") + raise NotThisMethod("unexpanded keywords, not a git-archive tarball") + refs = set([r.strip() for r in refnames.strip("()").split(",")]) + # starting in git-1.8.3, tags are listed as "tag: foo-1.0" instead of + # just "foo-1.0". If we see a "tag: " prefix, prefer those. + TAG = "tag: " + tags = set([r[len(TAG):] for r in refs if r.startswith(TAG)]) + if not tags: + # Either we're using git < 1.8.3, or there really are no tags. We use + # a heuristic: assume all version tags have a digit. The old git %d + # expansion behaves like git log --decorate=short and strips out the + # refs/heads/ and refs/tags/ prefixes that would let us distinguish + # between branches and tags. By ignoring refnames without digits, we + # filter out many common branch names like "release" and + # "stabilization", as well as "HEAD" and "master". + tags = set([r for r in refs if re.search(r'\d', r)]) + if verbose: + print("discarding '%s', no digits" % ",".join(refs-tags)) + if verbose: + print("likely tags: %s" % ",".join(sorted(tags))) + for ref in sorted(tags): + # sorting will prefer e.g. "2.0" over "2.0rc1" + if ref.startswith(tag_prefix): + r = ref[len(tag_prefix):] + if verbose: + print("picking %s" % r) + return {"version": r, + "full-revisionid": keywords["full"].strip(), + "dirty": False, "error": None + } + # no suitable tags, so version is "0+unknown", but full hex is still there + if verbose: + print("no suitable tags, using unknown + full revision id") + return {"version": "0+unknown", + "full-revisionid": keywords["full"].strip(), + "dirty": False, "error": "no suitable tags"} + + +@register_vcs_handler("git", "pieces_from_vcs") +def git_pieces_from_vcs(tag_prefix, root, verbose, run_command=run_command): + """Get version from 'git describe' in the root of the source tree. + + This only gets called if the git-archive 'subst' keywords were *not* + expanded, and _version.py hasn't already been rewritten with a short + version string, meaning we're inside a checked out source tree. + """ + if not os.path.exists(os.path.join(root, ".git")): + if verbose: + print("no .git in %s" % root) + raise NotThisMethod("no .git directory") + + GITS = ["git"] + if sys.platform == "win32": + GITS = ["git.cmd", "git.exe"] + # if there is a tag matching tag_prefix, this yields TAG-NUM-gHEX[-dirty] + # if there isn't one, this yields HEX[-dirty] (no NUM) + describe_out = run_command(GITS, ["describe", "--tags", "--dirty", + "--always", "--long", + "--match", "%s*" % tag_prefix], + cwd=root) + # --long was added in git-1.5.5 + if describe_out is None: + raise NotThisMethod("'git describe' failed") + describe_out = describe_out.strip() + full_out = run_command(GITS, ["rev-parse", "HEAD"], cwd=root) + if full_out is None: + raise NotThisMethod("'git rev-parse' failed") + full_out = full_out.strip() + + pieces = {} + pieces["long"] = full_out + pieces["short"] = full_out[:7] # maybe improved later + pieces["error"] = None + + # parse describe_out. It will be like TAG-NUM-gHEX[-dirty] or HEX[-dirty] + # TAG might have hyphens. + git_describe = describe_out + + # look for -dirty suffix + dirty = git_describe.endswith("-dirty") + pieces["dirty"] = dirty + if dirty: + git_describe = git_describe[:git_describe.rindex("-dirty")] + + # now we have TAG-NUM-gHEX or HEX + + if "-" in git_describe: + # TAG-NUM-gHEX + mo = re.search(r'^(.+)-(\d+)-g([0-9a-f]+)$', git_describe) + if not mo: + # unparseable. Maybe git-describe is misbehaving? + pieces["error"] = ("unable to parse git-describe output: '%s'" + % describe_out) + return pieces + + # tag + full_tag = mo.group(1) + if not full_tag.startswith(tag_prefix): + if verbose: + fmt = "tag '%s' doesn't start with prefix '%s'" + print(fmt % (full_tag, tag_prefix)) + pieces["error"] = ("tag '%s' doesn't start with prefix '%s'" + % (full_tag, tag_prefix)) + return pieces + pieces["closest-tag"] = full_tag[len(tag_prefix):] + + # distance: number of commits since tag + pieces["distance"] = int(mo.group(2)) + + # commit: short hex revision ID + pieces["short"] = mo.group(3) + + else: + # HEX: no tags + pieces["closest-tag"] = None + count_out = run_command(GITS, ["rev-list", "HEAD", "--count"], + cwd=root) + pieces["distance"] = int(count_out) # total number of commits + + return pieces + + +def do_vcs_install(manifest_in, versionfile_source, ipy): + """Git-specific installation logic for Versioneer. + + For Git, this means creating/changing .gitattributes to mark _version.py + for export-time keyword substitution. + """ + GITS = ["git"] + if sys.platform == "win32": + GITS = ["git.cmd", "git.exe"] + files = [manifest_in, versionfile_source] + if ipy: + files.append(ipy) + try: + me = __file__ + if me.endswith(".pyc") or me.endswith(".pyo"): + me = os.path.splitext(me)[0] + ".py" + versioneer_file = os.path.relpath(me) + except NameError: + versioneer_file = "versioneer.py" + files.append(versioneer_file) + present = False + try: + f = open(".gitattributes", "r") + for line in f.readlines(): + if line.strip().startswith(versionfile_source): + if "export-subst" in line.strip().split()[1:]: + present = True + f.close() + except EnvironmentError: + pass + if not present: + f = open(".gitattributes", "a+") + f.write("%s export-subst\n" % versionfile_source) + f.close() + files.append(".gitattributes") + run_command(GITS, ["add", "--"] + files) + + +def versions_from_parentdir(parentdir_prefix, root, verbose): + """Try to determine the version from the parent directory name. + + Source tarballs conventionally unpack into a directory that includes + both the project name and a version string. + """ + dirname = os.path.basename(root) + if not dirname.startswith(parentdir_prefix): + if verbose: + print("guessing rootdir is '%s', but '%s' doesn't start with " + "prefix '%s'" % (root, dirname, parentdir_prefix)) + raise NotThisMethod("rootdir doesn't start with parentdir_prefix") + return {"version": dirname[len(parentdir_prefix):], + "full-revisionid": None, + "dirty": False, "error": None} + +SHORT_VERSION_PY = """ +# This file was generated by 'versioneer.py' (0.16) from +# revision-control system data, or from the parent directory name of an +# unpacked source archive. Distribution tarballs contain a pre-generated copy +# of this file. + +import json +import sys + +version_json = ''' +%s +''' # END VERSION_JSON + + +def get_versions(): + return json.loads(version_json) +""" + + +def versions_from_file(filename): + """Try to determine the version from _version.py if present.""" + try: + with open(filename) as f: + contents = f.read() + except EnvironmentError: + raise NotThisMethod("unable to read _version.py") + mo = re.search(r"version_json = '''\n(.*)''' # END VERSION_JSON", + contents, re.M | re.S) + if not mo: + raise NotThisMethod("no version_json in _version.py") + return json.loads(mo.group(1)) + + +def write_to_version_file(filename, versions): + """Write the given version number to the given _version.py file.""" + os.unlink(filename) + contents = json.dumps(versions, sort_keys=True, + indent=1, separators=(",", ": ")) + with open(filename, "w") as f: + f.write(SHORT_VERSION_PY % contents) + + print("set %s to '%s'" % (filename, versions["version"])) + + +def plus_or_dot(pieces): + """Return a + if we don't already have one, else return a .""" + if "+" in pieces.get("closest-tag", ""): + return "." + return "+" + + +def render_pep440(pieces): + """Build up version string, with post-release "local version identifier". + + Our goal: TAG[+DISTANCE.gHEX[.dirty]] . Note that if you + get a tagged build and then dirty it, you'll get TAG+0.gHEX.dirty + + Exceptions: + 1: no tags. git_describe was just HEX. 0+untagged.DISTANCE.gHEX[.dirty] + """ + if pieces["closest-tag"]: + rendered = pieces["closest-tag"] + if pieces["distance"] or pieces["dirty"]: + rendered += plus_or_dot(pieces) + rendered += "%d.g%s" % (pieces["distance"], pieces["short"]) + if pieces["dirty"]: + rendered += ".dirty" + else: + # exception #1 + rendered = "0+untagged.%d.g%s" % (pieces["distance"], + pieces["short"]) + if pieces["dirty"]: + rendered += ".dirty" + return rendered + + +def render_pep440_pre(pieces): + """TAG[.post.devDISTANCE] -- No -dirty. + + Exceptions: + 1: no tags. 0.post.devDISTANCE + """ + if pieces["closest-tag"]: + rendered = pieces["closest-tag"] + if pieces["distance"]: + rendered += ".post.dev%d" % pieces["distance"] + else: + # exception #1 + rendered = "0.post.dev%d" % pieces["distance"] + return rendered + + +def render_pep440_post(pieces): + """TAG[.postDISTANCE[.dev0]+gHEX] . + + The ".dev0" means dirty. Note that .dev0 sorts backwards + (a dirty tree will appear "older" than the corresponding clean one), + but you shouldn't be releasing software with -dirty anyways. + + Exceptions: + 1: no tags. 0.postDISTANCE[.dev0] + """ + if pieces["closest-tag"]: + rendered = pieces["closest-tag"] + if pieces["distance"] or pieces["dirty"]: + rendered += ".post%d" % pieces["distance"] + if pieces["dirty"]: + rendered += ".dev0" + rendered += plus_or_dot(pieces) + rendered += "g%s" % pieces["short"] + else: + # exception #1 + rendered = "0.post%d" % pieces["distance"] + if pieces["dirty"]: + rendered += ".dev0" + rendered += "+g%s" % pieces["short"] + return rendered + + +def render_pep440_old(pieces): + """TAG[.postDISTANCE[.dev0]] . + + The ".dev0" means dirty. + + Eexceptions: + 1: no tags. 0.postDISTANCE[.dev0] + """ + if pieces["closest-tag"]: + rendered = pieces["closest-tag"] + if pieces["distance"] or pieces["dirty"]: + rendered += ".post%d" % pieces["distance"] + if pieces["dirty"]: + rendered += ".dev0" + else: + # exception #1 + rendered = "0.post%d" % pieces["distance"] + if pieces["dirty"]: + rendered += ".dev0" + return rendered + + +def render_git_describe(pieces): + """TAG[-DISTANCE-gHEX][-dirty]. + + Like 'git describe --tags --dirty --always'. + + Exceptions: + 1: no tags. HEX[-dirty] (note: no 'g' prefix) + """ + if pieces["closest-tag"]: + rendered = pieces["closest-tag"] + if pieces["distance"]: + rendered += "-%d-g%s" % (pieces["distance"], pieces["short"]) + else: + # exception #1 + rendered = pieces["short"] + if pieces["dirty"]: + rendered += "-dirty" + return rendered + + +def render_git_describe_long(pieces): + """TAG-DISTANCE-gHEX[-dirty]. + + Like 'git describe --tags --dirty --always -long'. + The distance/hash is unconditional. + + Exceptions: + 1: no tags. HEX[-dirty] (note: no 'g' prefix) + """ + if pieces["closest-tag"]: + rendered = pieces["closest-tag"] + rendered += "-%d-g%s" % (pieces["distance"], pieces["short"]) + else: + # exception #1 + rendered = pieces["short"] + if pieces["dirty"]: + rendered += "-dirty" + return rendered + + +def render(pieces, style): + """Render the given version pieces into the requested style.""" + if pieces["error"]: + return {"version": "unknown", + "full-revisionid": pieces.get("long"), + "dirty": None, + "error": pieces["error"]} + + if not style or style == "default": + style = "pep440" # the default + + if style == "pep440": + rendered = render_pep440(pieces) + elif style == "pep440-pre": + rendered = render_pep440_pre(pieces) + elif style == "pep440-post": + rendered = render_pep440_post(pieces) + elif style == "pep440-old": + rendered = render_pep440_old(pieces) + elif style == "git-describe": + rendered = render_git_describe(pieces) + elif style == "git-describe-long": + rendered = render_git_describe_long(pieces) + else: + raise ValueError("unknown style '%s'" % style) + + return {"version": rendered, "full-revisionid": pieces["long"], + "dirty": pieces["dirty"], "error": None} + + +class VersioneerBadRootError(Exception): + """The project root directory is unknown or missing key files.""" + + +def get_versions(verbose=False): + """Get the project version from whatever source is available. + + Returns dict with two keys: 'version' and 'full'. + """ + if "versioneer" in sys.modules: + # see the discussion in cmdclass.py:get_cmdclass() + del sys.modules["versioneer"] + + root = get_root() + cfg = get_config_from_root(root) + + assert cfg.VCS is not None, "please set [versioneer]VCS= in setup.cfg" + handlers = HANDLERS.get(cfg.VCS) + assert handlers, "unrecognized VCS '%s'" % cfg.VCS + verbose = verbose or cfg.verbose + assert cfg.versionfile_source is not None, \ + "please set versioneer.versionfile_source" + assert cfg.tag_prefix is not None, "please set versioneer.tag_prefix" + + versionfile_abs = os.path.join(root, cfg.versionfile_source) + + # extract version from first of: _version.py, VCS command (e.g. 'git + # describe'), parentdir. This is meant to work for developers using a + # source checkout, for users of a tarball created by 'setup.py sdist', + # and for users of a tarball/zipball created by 'git archive' or github's + # download-from-tag feature or the equivalent in other VCSes. + + get_keywords_f = handlers.get("get_keywords") + from_keywords_f = handlers.get("keywords") + if get_keywords_f and from_keywords_f: + try: + keywords = get_keywords_f(versionfile_abs) + ver = from_keywords_f(keywords, cfg.tag_prefix, verbose) + if verbose: + print("got version from expanded keyword %s" % ver) + return ver + except NotThisMethod: + pass + + try: + ver = versions_from_file(versionfile_abs) + if verbose: + print("got version from file %s %s" % (versionfile_abs, ver)) + return ver + except NotThisMethod: + pass + + from_vcs_f = handlers.get("pieces_from_vcs") + if from_vcs_f: + try: + pieces = from_vcs_f(cfg.tag_prefix, root, verbose) + ver = render(pieces, cfg.style) + if verbose: + print("got version from VCS %s" % ver) + return ver + except NotThisMethod: + pass + + try: + if cfg.parentdir_prefix: + ver = versions_from_parentdir(cfg.parentdir_prefix, root, verbose) + if verbose: + print("got version from parentdir %s" % ver) + return ver + except NotThisMethod: + pass + + if verbose: + print("unable to compute version") + + return {"version": "0+unknown", "full-revisionid": None, + "dirty": None, "error": "unable to compute version"} + + +def get_version(): + """Get the short version string for this project.""" + return get_versions()["version"] + + +def get_cmdclass(): + """Get the custom setuptools/distutils subclasses used by Versioneer.""" + if "versioneer" in sys.modules: + del sys.modules["versioneer"] + # this fixes the "python setup.py develop" case (also 'install' and + # 'easy_install .'), in which subdependencies of the main project are + # built (using setup.py bdist_egg) in the same python process. Assume + # a main project A and a dependency B, which use different versions + # of Versioneer. A's setup.py imports A's Versioneer, leaving it in + # sys.modules by the time B's setup.py is executed, causing B to run + # with the wrong versioneer. Setuptools wraps the sub-dep builds in a + # sandbox that restores sys.modules to it's pre-build state, so the + # parent is protected against the child's "import versioneer". By + # removing ourselves from sys.modules here, before the child build + # happens, we protect the child from the parent's versioneer too. + # Also see https://github.com/warner/python-versioneer/issues/52 + + cmds = {} + + # we add "version" to both distutils and setuptools + from distutils.core import Command + + class cmd_version(Command): + description = "report generated version string" + user_options = [] + boolean_options = [] + + def initialize_options(self): + pass + + def finalize_options(self): + pass + + def run(self): + vers = get_versions(verbose=True) + print("Version: %s" % vers["version"]) + print(" full-revisionid: %s" % vers.get("full-revisionid")) + print(" dirty: %s" % vers.get("dirty")) + if vers["error"]: + print(" error: %s" % vers["error"]) + cmds["version"] = cmd_version + + # we override "build_py" in both distutils and setuptools + # + # most invocation pathways end up running build_py: + # distutils/build -> build_py + # distutils/install -> distutils/build ->.. + # setuptools/bdist_wheel -> distutils/install ->.. + # setuptools/bdist_egg -> distutils/install_lib -> build_py + # setuptools/install -> bdist_egg ->.. + # setuptools/develop -> ? + + # we override different "build_py" commands for both environments + if "setuptools" in sys.modules: + from setuptools.command.build_py import build_py as _build_py + else: + from distutils.command.build_py import build_py as _build_py + + class cmd_build_py(_build_py): + def run(self): + root = get_root() + cfg = get_config_from_root(root) + versions = get_versions() + _build_py.run(self) + # now locate _version.py in the new build/ directory and replace + # it with an updated value + if cfg.versionfile_build: + target_versionfile = os.path.join(self.build_lib, + cfg.versionfile_build) + print("UPDATING %s" % target_versionfile) + write_to_version_file(target_versionfile, versions) + cmds["build_py"] = cmd_build_py + + if "cx_Freeze" in sys.modules: # cx_freeze enabled? + from cx_Freeze.dist import build_exe as _build_exe + + class cmd_build_exe(_build_exe): + def run(self): + root = get_root() + cfg = get_config_from_root(root) + versions = get_versions() + target_versionfile = cfg.versionfile_source + print("UPDATING %s" % target_versionfile) + write_to_version_file(target_versionfile, versions) + + _build_exe.run(self) + os.unlink(target_versionfile) + with open(cfg.versionfile_source, "w") as f: + LONG = LONG_VERSION_PY[cfg.VCS] + f.write(LONG % + {"DOLLAR": "$", + "STYLE": cfg.style, + "TAG_PREFIX": cfg.tag_prefix, + "PARENTDIR_PREFIX": cfg.parentdir_prefix, + "VERSIONFILE_SOURCE": cfg.versionfile_source, + }) + cmds["build_exe"] = cmd_build_exe + del cmds["build_py"] + + # we override different "sdist" commands for both environments + if "setuptools" in sys.modules: + from setuptools.command.sdist import sdist as _sdist + else: + from distutils.command.sdist import sdist as _sdist + + class cmd_sdist(_sdist): + def run(self): + versions = get_versions() + self._versioneer_generated_versions = versions + # unless we update this, the command will keep using the old + # version + self.distribution.metadata.version = versions["version"] + return _sdist.run(self) + + def make_release_tree(self, base_dir, files): + root = get_root() + cfg = get_config_from_root(root) + _sdist.make_release_tree(self, base_dir, files) + # now locate _version.py in the new base_dir directory + # (remembering that it may be a hardlink) and replace it with an + # updated value + target_versionfile = os.path.join(base_dir, cfg.versionfile_source) + print("UPDATING %s" % target_versionfile) + write_to_version_file(target_versionfile, + self._versioneer_generated_versions) + cmds["sdist"] = cmd_sdist + + return cmds + + +CONFIG_ERROR = """ +setup.cfg is missing the necessary Versioneer configuration. You need +a section like: + + [versioneer] + VCS = git + style = pep440 + versionfile_source = src/myproject/_version.py + versionfile_build = myproject/_version.py + tag_prefix = + parentdir_prefix = myproject- + +You will also need to edit your setup.py to use the results: + + import versioneer + setup(version=versioneer.get_version(), + cmdclass=versioneer.get_cmdclass(), ...) + +Please read the docstring in ./versioneer.py for configuration instructions, +edit setup.cfg, and re-run the installer or 'python versioneer.py setup'. +""" + +SAMPLE_CONFIG = """ +# See the docstring in versioneer.py for instructions. Note that you must +# re-run 'versioneer.py setup' after changing this section, and commit the +# resulting files. + +[versioneer] +#VCS = git +#style = pep440 +#versionfile_source = +#versionfile_build = +#tag_prefix = +#parentdir_prefix = + +""" + +INIT_PY_SNIPPET = """ +from ._version import get_versions +__version__ = get_versions()['version'] +del get_versions +""" + + +def do_setup(): + """Main VCS-independent setup function for installing Versioneer.""" + root = get_root() + try: + cfg = get_config_from_root(root) + except (EnvironmentError, configparser.NoSectionError, + configparser.NoOptionError) as e: + if isinstance(e, (EnvironmentError, configparser.NoSectionError)): + print("Adding sample versioneer config to setup.cfg", + file=sys.stderr) + with open(os.path.join(root, "setup.cfg"), "a") as f: + f.write(SAMPLE_CONFIG) + print(CONFIG_ERROR, file=sys.stderr) + return 1 + + print(" creating %s" % cfg.versionfile_source) + with open(cfg.versionfile_source, "w") as f: + LONG = LONG_VERSION_PY[cfg.VCS] + f.write(LONG % {"DOLLAR": "$", + "STYLE": cfg.style, + "TAG_PREFIX": cfg.tag_prefix, + "PARENTDIR_PREFIX": cfg.parentdir_prefix, + "VERSIONFILE_SOURCE": cfg.versionfile_source, + }) + + ipy = os.path.join(os.path.dirname(cfg.versionfile_source), + "__init__.py") + if os.path.exists(ipy): + try: + with open(ipy, "r") as f: + old = f.read() + except EnvironmentError: + old = "" + if INIT_PY_SNIPPET not in old: + print(" appending to %s" % ipy) + with open(ipy, "a") as f: + f.write(INIT_PY_SNIPPET) + else: + print(" %s unmodified" % ipy) + else: + print(" %s doesn't exist, ok" % ipy) + ipy = None + + # Make sure both the top-level "versioneer.py" and versionfile_source + # (PKG/_version.py, used by runtime code) are in MANIFEST.in, so + # they'll be copied into source distributions. Pip won't be able to + # install the package without this. + manifest_in = os.path.join(root, "MANIFEST.in") + simple_includes = set() + try: + with open(manifest_in, "r") as f: + for line in f: + if line.startswith("include "): + for include in line.split()[1:]: + simple_includes.add(include) + except EnvironmentError: + pass + # That doesn't cover everything MANIFEST.in can do + # (http://docs.python.org/2/distutils/sourcedist.html#commands), so + # it might give some false negatives. Appending redundant 'include' + # lines is safe, though. + if "versioneer.py" not in simple_includes: + print(" appending 'versioneer.py' to MANIFEST.in") + with open(manifest_in, "a") as f: + f.write("include versioneer.py\n") + else: + print(" 'versioneer.py' already in MANIFEST.in") + if cfg.versionfile_source not in simple_includes: + print(" appending versionfile_source ('%s') to MANIFEST.in" % + cfg.versionfile_source) + with open(manifest_in, "a") as f: + f.write("include %s\n" % cfg.versionfile_source) + else: + print(" versionfile_source already in MANIFEST.in") + + # Make VCS-specific changes. For git, this means creating/changing + # .gitattributes to mark _version.py for export-time keyword + # substitution. + do_vcs_install(manifest_in, cfg.versionfile_source, ipy) + return 0 + + +def scan_setup_py(): + """Validate the contents of setup.py against Versioneer's expectations.""" + found = set() + setters = False + errors = 0 + with open("setup.py", "r") as f: + for line in f.readlines(): + if "import versioneer" in line: + found.add("import") + if "versioneer.get_cmdclass()" in line: + found.add("cmdclass") + if "versioneer.get_version()" in line: + found.add("get_version") + if "versioneer.VCS" in line: + setters = True + if "versioneer.versionfile_source" in line: + setters = True + if len(found) != 3: + print("") + print("Your setup.py appears to be missing some important items") + print("(but I might be wrong). Please make sure it has something") + print("roughly like the following:") + print("") + print(" import versioneer") + print(" setup( version=versioneer.get_version(),") + print(" cmdclass=versioneer.get_cmdclass(), ...)") + print("") + errors += 1 + if setters: + print("You should remove lines like 'versioneer.VCS = ' and") + print("'versioneer.versionfile_source = ' . This configuration") + print("now lives in setup.cfg, and should be removed from setup.py") + print("") + errors += 1 + return errors + +if __name__ == "__main__": + cmd = sys.argv[1] + if cmd == "setup": + errors = do_setup() + errors += scan_setup_py() + if errors: + sys.exit(1) From 9344ddb7acca25605ec9f1857879f867b88a5df9 Mon Sep 17 00:00:00 2001 From: Joris Van den Bossche Date: Tue, 31 May 2016 12:58:52 +0200 Subject: [PATCH 17/34] DOC: add rtree and pysal to doc requirements (needed for new mapping and merging docs) (#332) --- doc/environment.yml | 2 ++ 1 file changed, 2 insertions(+) diff --git a/doc/environment.yml b/doc/environment.yml index 1c81308..0daf4ef 100644 --- a/doc/environment.yml +++ b/doc/environment.yml @@ -7,10 +7,12 @@ dependencies: - shapely - fiona - pyproj +- rtree - six - geopy - matplotlib - descartes +- pysal - sphinx - sphinx_rtd_theme - ipython=4.0.1 From dc4b5486960f9d080a85d8489e217eac4fadf43f Mon Sep 17 00:00:00 2001 From: Nick Eubank Date: Tue, 31 May 2016 07:56:17 -0700 Subject: [PATCH 18/34] final doc updates (#328) --- doc/source/aggregation_with_dissolve.rst | 51 ++++++++++++++++++++++++ doc/source/geometric_manipulations.rst | 25 +++++------- doc/source/index.rst | 1 + doc/source/mergingdata.rst | 4 +- doc/source/set_operations.rst | 5 +-- 5 files changed, 67 insertions(+), 19 deletions(-) create mode 100644 doc/source/aggregation_with_dissolve.rst diff --git a/doc/source/aggregation_with_dissolve.rst b/doc/source/aggregation_with_dissolve.rst new file mode 100644 index 0000000..a191fb5 --- /dev/null +++ b/doc/source/aggregation_with_dissolve.rst @@ -0,0 +1,51 @@ +.. ipython:: python + :suppress: + + import geopandas as gpd + + +Aggregation with dissolve +============================= + +It is often the case that we find ourselves working with spatial data that is more granular than we need. For example, we might have data on sub-national units, but we're actually interested in studying patterns at the level of countries. + +In a non-spatial setting, we aggregate our data using the ``groupby`` function. But when working with spatial data, we need a special tool that can also aggregate geometric features. In the *geopandas* library, that functionality is provided by the ``dissolve`` function. + +``dissolve`` can be thought of as doing three things: (a) it dissolves all the geometries within a given group together into a single geometric feature (using the ``unary_union`` method), and (b) it aggregates all the rows of data in a group using ``groupby.aggregate()``, and (c) it combines those two results. + +``dissolve`` Example +~~~~~~~~~~~~~~~~~~~~~ + +Suppose we are interested in studying continents, but we only have country-level data like the country dataset included in *geopandas*. We can easily convert this to a continent-level dataset. + + +First, let's look at the most simple case where we just want continent shapes and names. By default, ``dissolve`` will pass ``'first'`` to ``groupby.aggregate``. + +.. ipython:: python + + world = gpd.read_file(gpd.datasets.get_path('naturalearth_lowres')) + world = world[['continent', 'geometry']] + continents = world.dissolve(by='continent') + + @savefig continents.png width=5in + continents.plot(); + + continents.head() + +If we are interested in aggregate populations, however, we can pass different functions to the ``dissolve`` method to aggregate populations: + +.. ipython:: python + + world = gpd.read_file(gpd.datasets.get_path('naturalearth_lowres')) + world = world[['continent', 'geometry', 'pop_est']] + continents = world.dissolve(by='continent', aggfunc='sum') + + @savefig continents.png width=5in + continents.plot(column = 'pop_est', scheme='quantiles', cmap='YlOrRd'); + + continents.head() + + + +.. toctree:: + :maxdepth: 2 diff --git a/doc/source/geometric_manipulations.rst b/doc/source/geometric_manipulations.rst index 29f6613..b1b256c 100644 --- a/doc/source/geometric_manipulations.rst +++ b/doc/source/geometric_manipulations.rst @@ -1,7 +1,7 @@ Geometric Manipulations ======================== -*geopandas* makes available all the tools for geometric manipulations in the `*shapely* library `_. +*geopandas* makes available all the tools for geometric manipulations in the `*shapely* library `_. Note that documentation for all set-theoretic tools for creating new shapes using the relationship between two different spatial datasets -- like creating intersections, or differences -- can be found on the :doc:`set operations ` page. @@ -40,6 +40,11 @@ Constructive Methods Returns a ``GeoSeries`` containing a simplified representation of each object. +.. attribute:: GeoSeries.unary_union + + Return a geometry containing the union of all geometries in the ``GeoSeries``. + + Affine transformations ~~~~~~~~~~~~~~~~~~~~~~~~ @@ -59,16 +64,10 @@ Affine transformations Shift the coordinates of the GeoSeries. -Aggregation Methods -~~~~~~~~~~~~~~~~~~~~ - -.. attribute:: GeoSeries.unary_union - - Return a geometry containing the union of all geometries in the ``GeoSeries``. Examples of Geometric Manipulations ------------------------------------- +------------------------------------ .. sourcecode:: python @@ -128,7 +127,7 @@ GeoPandas also implements alternate constructors that can read any data format r 3 Brooklyn 1.959432e+09 726568.946340 4 Queens 3.049947e+09 861038.479299 5 Staten Island 1.623853e+09 330385.036974 - + geometry BoroCode 1 (POLYGON ((981219.0557861328125000 188655.3157... @@ -138,7 +137,7 @@ GeoPandas also implements alternate constructors that can read any data format r 5 (POLYGON ((970217.0223999023437500 145643.3322... .. image:: _static/nyc.png - + .. sourcecode:: python >>> boros['geometry'].convex_hull @@ -183,7 +182,7 @@ just use: >>> holes = boros['geometry'].intersection(mp) .. image:: _static/holes.png - + and to get the area outside of the holes: .. sourcecode:: python @@ -191,7 +190,7 @@ and to get the area outside of the holes: >>> boros_with_holes = boros['geometry'].difference(mp) .. image:: _static/boros_with_holes.png - + Note that this can be simplified a bit, since ``geometry`` is available as an attribute on a ``GeoDataFrame``, and the ``intersection`` and ``difference`` methods are implemented with the @@ -221,5 +220,3 @@ borough that are in the holes: .. toctree:: :maxdepth: 2 - - diff --git a/doc/source/index.rst b/doc/source/index.rst index b380c35..0395b88 100644 --- a/doc/source/index.rst +++ b/doc/source/index.rst @@ -33,6 +33,7 @@ such as PostGIS. Managing Projections Geometric Manipulations Set Operations with overlay + Aggregation with dissolve Merging Data Geocoding Reference to All Attributes and Methods diff --git a/doc/source/mergingdata.rst b/doc/source/mergingdata.rst index ce75ff2..3ebd69c 100644 --- a/doc/source/mergingdata.rst +++ b/doc/source/mergingdata.rst @@ -67,8 +67,8 @@ In a Spatial Join, two geometry objects are merged based on their spatial relati cities.head() # Execute spatial join - from geopandas.tools import sjoin - cities_with_country = sjoin(cities, countries, how="inner", op='intersects') + + cities_with_country = gpd.sjoin(cities, countries, how="inner", op='intersects') cities_with_country.head() diff --git a/doc/source/set_operations.rst b/doc/source/set_operations.rst index 2ffc57c..69d7f29 100644 --- a/doc/source/set_operations.rst +++ b/doc/source/set_operations.rst @@ -57,8 +57,7 @@ To select only the portion of countries within 500km of a capital, we specify th .. ipython:: python - from geopandas.tools import overlay - country_cores = overlay(countries, capitals, how='intersection') + country_cores = gpd.overlay(countries, capitals, how='intersection') @savefig country_cores.png width=5in country_cores.plot(); @@ -66,7 +65,7 @@ Changing the "how" option allows for different types of overlay operations. For .. ipython:: python - country_peripheries = overlay(countries, capitals, how='difference') + country_peripheries = gpd.overlay(countries, capitals, how='difference') @savefig country_peripheries.png width=5in country_peripheries.plot(); From 035c3147516b135e9b86ecb7576426b37044b0b1 Mon Sep 17 00:00:00 2001 From: Joris Van den Bossche Date: Mon, 6 Jun 2016 01:00:19 +0200 Subject: [PATCH 19/34] TST: explicitly skip tests that fail with master version of matplotlib (#327) --- geopandas/tests/test_plotting.py | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/geopandas/tests/test_plotting.py b/geopandas/tests/test_plotting.py index c0a614c..ac6a468 100644 --- a/geopandas/tests/test_plotting.py +++ b/geopandas/tests/test_plotting.py @@ -4,6 +4,7 @@ import numpy as np import os import shutil import tempfile +from distutils.version import LooseVersion import matplotlib matplotlib.use('Agg', warn=False) @@ -24,7 +25,7 @@ GENERATE_BASELINE = False BASELINE_DIR = os.path.join(os.path.dirname(__file__), 'baseline_images', 'test_plotting') TRAVIS = bool(os.environ.get('TRAVIS', False)) - +MPL_DEV = matplotlib.__version__ > LooseVersion('1.5.1') class TestImageComparisons(unittest.TestCase): @@ -52,6 +53,7 @@ class TestImageComparisons(unittest.TestCase): 'vs. %(expected)s ' '(RMS %(rms).3f)' % err) + @unittest.skipIf(MPL_DEV, 'Skip for development version of matplotlib') def test_poly_plot(self): """ Test plotting a simple series of polygons """ clf() @@ -62,6 +64,7 @@ class TestImageComparisons(unittest.TestCase): ax = polys.plot() self._compare_images(ax=ax, filename=filename) + @unittest.skipIf(MPL_DEV, 'Skip for development version of matplotlib') def test_point_plot(self): """ Test plotting a simple series of points """ clf() @@ -71,6 +74,7 @@ class TestImageComparisons(unittest.TestCase): ax = points.plot() self._compare_images(ax=ax, filename=filename) + @unittest.skipIf(MPL_DEV, 'Skip for development version of matplotlib') def test_line_plot(self): """ Test plotting a simple series of lines """ clf() @@ -119,6 +123,7 @@ class TestPointPlotting(unittest.TestCase): values = np.arange(self.N) self.df = GeoDataFrame({'geometry': self.points, 'values': values}) + @unittest.skipIf(MPL_DEV, 'Skip for development version of matplotlib') def test_default_colors(self): ## without specifying values -> max 9 different colors From 752ab5ddb703eaf6cc8148ea7d588f1ee4499325 Mon Sep 17 00:00:00 2001 From: Joris Van den Bossche Date: Mon, 6 Jun 2016 23:52:18 +0200 Subject: [PATCH 20/34] DOC: update docstrings of overlay and sjoin (#336) --- geopandas/tools/overlay.py | 22 ++++++++++++++-------- geopandas/tools/sjoin.py | 31 +++++++++++++++++++------------ 2 files changed, 33 insertions(+), 20 deletions(-) diff --git a/geopandas/tools/overlay.py b/geopandas/tools/overlay.py index d19a13b..8abbca7 100644 --- a/geopandas/tools/overlay.py +++ b/geopandas/tools/overlay.py @@ -17,7 +17,7 @@ def _uniquify(columns): def _extract_rings(df): - """Collects all inner and outer linear rings from a GeoDataFrame + """Collects all inner and outer linear rings from a GeoDataFrame with (multi)Polygon geometeries Parameters @@ -55,22 +55,28 @@ def _extract_rings(df): return rings def overlay(df1, df2, how, use_sindex=True): - """Perform spatial overlay between two polygons - Currently only supports data GeoDataFrames with polygons + """Perform spatial overlay between two polygons. - Implements several methods (see `allowed_hows` list) that are - all effectively subsets of the union. + Currently only supports data GeoDataFrames with polygons. + Implements several methods that are all effectively subsets of + the union. Parameters ---------- df1 : GeoDataFrame with MultiPolygon or Polygon geometry column df2 : GeoDataFrame with MultiPolygon or Polygon geometry column - how : method of spatial overlay - use_sindex : Boolean; Use the spatial index to speed up operation. Default is True. + how : string + Method of spatial overlay: 'intersection', 'union', + 'identity', 'symmetric_difference' or 'difference'. + use_sindex : boolean, default True + Use the spatial index to speed up operation if available. Returns ------- - df : GeoDataFrame with new set of polygons and attributes resulting from the overlay + df : GeoDataFrame + GeoDataFrame with new set of polygons and attributes + resulting from the overlay + """ allowed_hows = [ 'intersection', diff --git a/geopandas/tools/sjoin.py b/geopandas/tools/sjoin.py index ba6c7cc..206f4cb 100644 --- a/geopandas/tools/sjoin.py +++ b/geopandas/tools/sjoin.py @@ -4,21 +4,28 @@ from shapely import prepared def sjoin(left_df, right_df, how='inner', op='intersects', - lsuffix='left', rsuffix='right', **kwargs): + lsuffix='left', rsuffix='right'): """Spatial join of two GeoDataFrames. - left_df, right_df are GeoDataFrames - how: type of join - left -> use keys from left_df; retain only left_df geometry column - right -> use keys from right_df; retain only right_df geometry column - inner -> use intersection of keys from both dfs; - retain only left_df geometry column - op: binary predicate {'intersects', 'contains', 'within'} - see http://toblerity.org/shapely/manual.html#binary-predicates - lsuffix: suffix to apply to overlapping column names (left GeoDataFrame) - rsuffix: suffix to apply to overlapping column names (right GeoDataFrame) - """ + Parameters + ---------- + left_df, right_df : GeoDataFrames + how : string, default 'inner' + The type of join: + * 'left': use keys from left_df; retain only left_df geometry column + * 'right': use keys from right_df; retain only right_df geometry column + * 'inner': use intersection of keys from both dfs; retain only + left_df geometry column + op : string, default 'intersection' + Binary predicate, one of {'intersects', 'contains', 'within'}. + See http://toblerity.org/shapely/manual.html#binary-predicates. + lsuffix : string, default 'left' + Suffix to apply to overlapping column names (left GeoDataFrame). + rsuffix : string, default 'right' + Suffix to apply to overlapping column names (right GeoDataFrame). + + """ import rtree allowed_hows = ['left', 'right', 'inner'] From 7a3d28784f3e37dbceefbe99e03c10a8db220b63 Mon Sep 17 00:00:00 2001 From: Joris Van den Bossche Date: Wed, 8 Jun 2016 13:24:03 +0200 Subject: [PATCH 21/34] Add sindex / cx to changelog --- CHANGELOG | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/CHANGELOG b/CHANGELOG index a7476bb..58d7c8a 100644 --- a/CHANGELOG +++ b/CHANGELOG @@ -10,6 +10,11 @@ Improvements: to represent the ``GeoSeries`` as a GeoJSON-like ``FeatureCollection`` (#116) and ``iterfeatures`` method (#178) * Addition of the ``explode`` (#146) and ``dissolve`` (#310, #311) methods. +* Addition of the ``sindex`` attribute, a Spatial Index using the optional + dependency ``rtree`` (``libspatialindex``) that can be used to speed up + certain operations such as overlays (#140, #141). +* Addition of the ``GeoSeries.ix`` coordinate indexer to slice a GeoSeries based + on a bounding box of the coordinates (#55). * Improvements to plotting: ability to specify edge colors (#173), support for the ``vmin``, ``vmax``, ``figsize``, ``linewidth`` keywords (#207), legends for chloropleth plots (#210), color points by specifying a colormap (#186) or From 6bda8e187f4440abe96a9650f8e35647fc72c2c7 Mon Sep 17 00:00:00 2001 From: Joris Van den Bossche Date: Wed, 8 Jun 2016 14:24:16 +0200 Subject: [PATCH 22/34] DOC: add overview of overlay methods (#337) --- doc/source/set_operations.rst | 137 +++++++++++++++++++++++++++++++--- 1 file changed, 126 insertions(+), 11 deletions(-) diff --git a/doc/source/set_operations.rst b/doc/source/set_operations.rst index 69d7f29..2db553b 100644 --- a/doc/source/set_operations.rst +++ b/doc/source/set_operations.rst @@ -7,35 +7,147 @@ Set-Operations with Overlay ============================ -When working with multiple spatial datasets -- especially multiple *polygon* or *line* datasets -- users often wish to create new shapes based on places where those datasets overlap (or don't overlap). These manipulations are often referred using the language of sets -- intersections, unions, and differences. These types of operations are made available in the *geopandas* library through the ``overlay`` function. +When working with multiple spatial datasets -- especially multiple *polygon* or +*line* datasets -- users often wish to create new shapes based on places where +those datasets overlap (or don't overlap). These manipulations are often +referred using the language of sets -- intersections, unions, and differences. +These types of operations are made available in the *geopandas* library through +the ``overlay`` function. -The basic idea is demonstrated by the graphic below but keep in mind that overlays operate at the DataFrame level, not on individual geometries, and the properties from both are retained. In effect, for every shape in the first GeoDataFrame, this operation is executed against every other shape in the other GeoDataFrame: +The basic idea is demonstrated by the graphic below but keep in mind that +overlays operate at the DataFrame level, not on individual geometries, and the +properties from both are retained. In effect, for every shape in the first +GeoDataFrame, this operation is executed against every other shape in the other +GeoDataFrame: .. image:: _static/overlay_operations.png **Source: QGIS Documentation** -(Note to users familiar with the *shapely* library: ``overlay`` can be thought of as offering versions of the standard *shapely* set-operations that deal with the complexities of applying set operations to two *GeoSeries*. The standard *shapely* set-operations are also available as ``GeoSeries`` methods.) +(Note to users familiar with the *shapely* library: ``overlay`` can be thought +of as offering versions of the standard *shapely* set-operations that deal with +the complexities of applying set operations to two *GeoSeries*. The standard +*shapely* set-operations are also available as ``GeoSeries`` methods.) -Overlay Example ------------------ +The different Overlay operations +-------------------------------- -First, we load some example data: +First, we create some example data: + +.. ipython:: python + + from shapely.geometry import Polygon + polys1 = gpd.GeoSeries([Polygon([(0,0), (2,0), (2,2), (0,2)]), + Polygon([(2,2), (4,2), (4,4), (2,4)])]) + polys2 = gpd.GeoSeries([Polygon([(1,1), (3,1), (3,3), (1,3)]), + Polygon([(3,3), (5,3), (5,5), (3,5)])]) + + df1 = gpd.GeoDataFrame({'geometry': polys1, 'df1':[1,2]}) + df2 = gpd.GeoDataFrame({'geometry': polys2, 'df2':[1,2]}) + +These two GeoDataFrames have some overlapping areas: + +.. ipython:: python + + ax = df1.plot(color='red'); + @savefig overlay_example.png width=5in + df2.plot(ax=ax, color='green'); + +We illustrate the different overlay modes with the above example. +The ``overlay`` function will determine the set of all individual geometries +from overlaying the two input GeoDataFrames. This result covers the area covered +by the two input GeoDataFrames, and also preserves all unique regions defined by +the combined boundaries of the two GeoDataFrames. + +When using ``how='union'``, all those possible geometries are returned: + +.. ipython:: python + + res_union = gpd.overlay(df1, df2, how='union') + res_union + + ax = res_union.plot() + df1.plot(ax=ax, facecolor='none'); + @savefig overlay_example_union.png width=5in + df2.plot(ax=ax, facecolor='none'); + +The other ``how`` operations will return different subsets of those geometries. +With ``how='intersection'``, it returns only those geometries that are contained +by both GeoDataFrames: + +.. ipython:: python + + res_intersection = gpd.overlay(df1, df2, how='intersection') + res_intersection + + ax = res_intersection.plot() + df1.plot(ax=ax, facecolor='none'); + @savefig overlay_example_intersection.png width=5in + df2.plot(ax=ax, facecolor='none'); + +``how='symmetric_difference'`` is the opposite of ``'intersection'`` and returns +the geometries that are only part of one of the GeoDataFrames but not of both: + +.. ipython:: python + + res_symdiff = gpd.overlay(df1, df2, how='symmetric_difference') + res_symdiff + + ax = res_symdiff.plot() + df1.plot(ax=ax, facecolor='none'); + @savefig overlay_example_symdiff.png width=5in + df2.plot(ax=ax, facecolor='none'); + +To obtain the geometries that are part of ``df1`` but are not contained in +``df2``, you can use ``how='difference'``: + +.. ipython:: python + + res_difference = gpd.overlay(df1, df2, how='difference') + res_difference + + ax = res_difference.plot() + df1.plot(ax=ax, facecolor='none'); + @savefig overlay_example_difference.png width=5in + df2.plot(ax=ax, facecolor='none'); + +Finally, with ``how='identity'``, the result consists of the surface of ``df1``, +but with the geometries obtained from overlaying ``df1`` with ``df2``: + +.. ipython:: python + + res_identity = gpd.overlay(df1, df2, how='identity') + res_identity + + ax = res_identity.plot() + df1.plot(ax=ax, facecolor='none'); + @savefig overlay_example_identity.png width=5in + df2.plot(ax=ax, facecolor='none'); + + +Overlay Countries Example +------------------------- + +First, we load the countries and cities example datasets and select : .. ipython:: python world = gpd.read_file(gpd.datasets.get_path('naturalearth_lowres')) capitals = gpd.read_file(gpd.datasets.get_path('naturalearth_cities')) - # Select some columns - countries = world[['geometry', 'name']] + # Select South Amarica and some columns + countries = world[world['continent'] == "South America"] + countries = countries[['geometry', 'name']] # Project to crs that uses meters as distance measure - countries = countries.to_crs('+init=epsg:3395')[countries.name!="Antarctica"] + countries = countries.to_crs('+init=epsg:3395') capitals = capitals.to_crs('+init=epsg:3395') -To illustrate the ``overlay`` function, consider the following case in which one wishes to identify the "core" portion of each country -- defined as areas within 500km of a capital -- using a ``GeoDataFrame`` of countries and a ``GeoDataFrame`` of capitals. +To illustrate the ``overlay`` function, consider the following case in which one +wishes to identify the "core" portion of each country -- defined as areas within +500km of a capital -- using a ``GeoDataFrame`` of countries and a +``GeoDataFrame`` of capitals. .. ipython:: python @@ -69,8 +181,11 @@ Changing the "how" option allows for different types of overlay operations. For @savefig country_peripheries.png width=5in country_peripheries.plot(); + + + More Examples ------------------ +------------- A larger set of examples of the use of ``overlay`` can be found `here `_ From 35403c8fba75afb7c4f7ac9b4bf171cb7d139c49 Mon Sep 17 00:00:00 2001 From: Joris Van den Bossche Date: Wed, 8 Jun 2016 14:48:11 +0200 Subject: [PATCH 23/34] DOC: disable other formats on readthedocs (for faster build) (#340) --- readthedocs.yml | 2 ++ 1 file changed, 2 insertions(+) diff --git a/readthedocs.yml b/readthedocs.yml index 1c401c1..50b41dd 100644 --- a/readthedocs.yml +++ b/readthedocs.yml @@ -1,3 +1,5 @@ +formats: + - none conda: file: doc/environment.yml python: From 04e450634d3f776d4d3bdc6c117e5815f9375a3a Mon Sep 17 00:00:00 2001 From: Joris Van den Bossche Date: Wed, 8 Jun 2016 15:13:12 +0200 Subject: [PATCH 24/34] DOC: remove hardcoded version (leftover after #331) --- doc/source/conf.py | 9 ++------- 1 file changed, 2 insertions(+), 7 deletions(-) diff --git a/doc/source/conf.py b/doc/source/conf.py index c713508..ae2b50d 100644 --- a/doc/source/conf.py +++ b/doc/source/conf.py @@ -48,13 +48,8 @@ copyright = u'2013-2014, GeoPandas developers' # The version info for the project you're documenting, acts as replacement for # |version| and |release|, also used in various other places throughout the # built documents. -d = {} -try: - execfile(os.path.join('..', '..', 'geopandas', 'version.py'), d) - version = release = d['version'] -except: - # FIXME: This shouldn't be hardwired, but should be set one place only - version = release = '0.2.0.dev' +import geopandas +version = release = geopandas.__version__ # The language for content autogenerated by Sphinx. Refer to documentation # for a list of supported languages. From 75f74804ca54bd488058c73ca86b489251a52d1c Mon Sep 17 00:00:00 2001 From: Kelsey Jordahl Date: Fri, 10 Jun 2016 01:33:50 -0700 Subject: [PATCH 25/34] BUG: Use _geometry_column_name in set_geometry() (#342) * BUG: Use _geometry_column_name rather than default name * TST: Add a test for to_crs() with different geometry column name * TST: Remove assertions that no longer hold for set_geometry() * TST CLN: simpler to use rename() * TST: comment instead of docstring for test method --- geopandas/geodataframe.py | 4 ++-- geopandas/tests/test_geodataframe.py | 16 ++++++++++++---- 2 files changed, 14 insertions(+), 6 deletions(-) diff --git a/geopandas/geodataframe.py b/geopandas/geodataframe.py index 8c9e5b3..3f0817d 100644 --- a/geopandas/geodataframe.py +++ b/geopandas/geodataframe.py @@ -125,7 +125,7 @@ class GeoDataFrame(GeoPandasBase, DataFrame): crs = getattr(col, 'crs', self.crs) to_remove = None - geo_column_name = DEFAULT_GEO_COLUMN_NAME + geo_column_name = self._geometry_column_name if isinstance(col, (Series, list, np.ndarray)): level = col elif hasattr(col, 'ndim') and col.ndim != 1: @@ -139,7 +139,7 @@ class GeoDataFrame(GeoPandasBase, DataFrame): raise if drop: to_remove = col - geo_column_name = DEFAULT_GEO_COLUMN_NAME + geo_column_name = self._geometry_column_name else: geo_column_name = col diff --git a/geopandas/tests/test_geodataframe.py b/geopandas/tests/test_geodataframe.py index 2796841..6739cd2 100644 --- a/geopandas/tests/test_geodataframe.py +++ b/geopandas/tests/test_geodataframe.py @@ -55,16 +55,12 @@ class TestDataFrame(unittest.TestCase): geom2 = [Point(x, y) for x, y in zip(range(5, 10), range(5))] df2 = df.set_geometry(geom2, crs='dummy_crs') - self.assert_('geometry' in df2) self.assert_('location' in df2) self.assertEqual(df2.crs, 'dummy_crs') self.assertEqual(df2.geometry.crs, 'dummy_crs') # reset so it outputs okay df2.crs = df.crs assert_geoseries_equal(df2.geometry, GeoSeries(geom2, crs=df2.crs)) - # for right now, non-geometry comes back as series - assert_geoseries_equal(df2['location'], df['location'], - check_series_type=False, check_dtype=False) def test_geo_getitem(self): data = {"A": range(5), "B": range(-5, 0), @@ -372,6 +368,18 @@ class TestDataFrame(unittest.TestCase): utm = lonlat.to_crs(epsg=26918) self.assertTrue(all(df2['geometry'].geom_almost_equals(utm['geometry'], decimal=2))) + def test_to_crs_geo_column_name(self): + # Test to_crs() with different geometry column name (GH#339) + df2 = self.df2.copy() + df2.crs = {'init': 'epsg:26918', 'no_defs': True} + df2 = df2.rename(columns={'geometry': 'geom'}) + df2.set_geometry('geom', inplace=True) + lonlat = df2.to_crs(epsg=4326) + utm = lonlat.to_crs(epsg=26918) + self.assertEqual(lonlat.geometry.name, 'geom') + self.assertEqual(utm.geometry.name, 'geom') + self.assertTrue(all(df2.geometry.geom_almost_equals(utm.geometry, decimal=2))) + def test_from_features(self): nybb_filename, nybb_zip_path = download_nybb() with fiona.open(nybb_zip_path, From 8d6af00efd86b6106273006e37cda2a7d9ef436c Mon Sep 17 00:00:00 2001 From: Kelsey Jordahl Date: Fri, 10 Jun 2016 10:13:38 -0700 Subject: [PATCH 26/34] DOC: Remove TODO from README --- README.md | 7 ------- 1 file changed, 7 deletions(-) diff --git a/README.md b/README.md index 4a410bc..233fd71 100644 --- a/README.md +++ b/README.md @@ -113,10 +113,3 @@ GeoPandas also implements alternate constructors that can read any data format r dtype: object ![Convex hulls of New York City boroughs](examples/nyc_hull.png) - -TODO ----- - -- Finish implementing and testing pandas methods on GeoPandas objects -- The current GeoDataFrame does not do very much. -- spatial joins, grouping and more... From 420bbc85959dbba5d78e7d793e8de5f701074a6c Mon Sep 17 00:00:00 2001 From: Kelsey Jordahl Date: Fri, 10 Jun 2016 10:16:35 -0700 Subject: [PATCH 27/34] DOC: Add links to documentation in README --- README.md | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/README.md b/README.md index 233fd71..6c5f2ee 100644 --- a/README.md +++ b/README.md @@ -20,6 +20,11 @@ transformed to new coordinate systems with the `to_crs()` method. There is currently no enforcement of like coordinates for operations, but that may change in the future. +Documentation is available at [geopandas.org](http://geopandas.org) +(current release) and +[Read the Docs](http://geopandas.readthedocs.io/en/master/) +(development version). + Install -------- From 7fb7c458a9c53e86eead87302817def01c92764f Mon Sep 17 00:00:00 2001 From: Kelsey Jordahl Date: Fri, 10 Jun 2016 10:51:08 -0700 Subject: [PATCH 28/34] DOC: Remove outdated/redundant comment --- setup.py | 1 - 1 file changed, 1 deletion(-) diff --git a/setup.py b/setup.py index 1b63619..363c71e 100644 --- a/setup.py +++ b/setup.py @@ -1,7 +1,6 @@ #!/usr/bin/env/python """Installation script -Version handling borrowed from pandas project. """ import sys From c305a5aa375f06bdd34090cc5505495e24f5fad2 Mon Sep 17 00:00:00 2001 From: Kelsey Jordahl Date: Fri, 10 Jun 2016 10:51:18 -0700 Subject: [PATCH 29/34] CLN: Remove unused imports --- setup.py | 2 -- 1 file changed, 2 deletions(-) diff --git a/setup.py b/setup.py index 363c71e..f7461b1 100644 --- a/setup.py +++ b/setup.py @@ -3,9 +3,7 @@ """ -import sys import os -import warnings try: from setuptools import setup From 320f7299d76c993298f3634bebd04b324b76eb4d Mon Sep 17 00:00:00 2001 From: Kelsey Jordahl Date: Fri, 10 Jun 2016 10:52:43 -0700 Subject: [PATCH 30/34] BLD: Update authos and email --- setup.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/setup.py b/setup.py index f7461b1..3d2bed0 100644 --- a/setup.py +++ b/setup.py @@ -43,8 +43,8 @@ setup(name='geopandas', version=versioneer.get_version(), description='Geographic pandas extensions', license='BSD', - author='Kelsey Jordahl', - author_email='kjordahl@enthought.com', + author='GeoPandas contributors', + author_email='kjordahl@alum.mit.edu', url='http://geopandas.org', long_description=LONG_DESCRIPTION, packages=['geopandas', 'geopandas.io', 'geopandas.tools', From e22e9a5a1b5de1d7b937e2194030e0ede42d895c Mon Sep 17 00:00:00 2001 From: Kelsey Jordahl Date: Fri, 10 Jun 2016 10:53:22 -0700 Subject: [PATCH 31/34] Update copyright date --- LICENSE.txt | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/LICENSE.txt b/LICENSE.txt index 15d5aed..7eb8134 100644 --- a/LICENSE.txt +++ b/LICENSE.txt @@ -1,4 +1,4 @@ -Copyright (c) 2013, GeoPandas developers. +Copyright (c) 2013-2016, GeoPandas developers. All rights reserved. Redistribution and use in source and binary forms, with or without From 353f5c5322d1e46b80a6db31f4c97560642c60a3 Mon Sep 17 00:00:00 2001 From: Kelsey Jordahl Date: Fri, 10 Jun 2016 10:54:41 -0700 Subject: [PATCH 32/34] DOC: Clarify multiple version on RTD --- README.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/README.md b/README.md index 6c5f2ee..a1f04fb 100644 --- a/README.md +++ b/README.md @@ -23,7 +23,7 @@ but that may change in the future. Documentation is available at [geopandas.org](http://geopandas.org) (current release) and [Read the Docs](http://geopandas.readthedocs.io/en/master/) -(development version). +(release and development versions). Install -------- From 24bc3dc3a703827a1fa009295852b23523c15f4f Mon Sep 17 00:00:00 2001 From: Kelsey Jordahl Date: Fri, 10 Jun 2016 11:33:59 -0700 Subject: [PATCH 33/34] Update copyright in docs --- doc/source/conf.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/doc/source/conf.py b/doc/source/conf.py index ae2b50d..d563ab3 100644 --- a/doc/source/conf.py +++ b/doc/source/conf.py @@ -43,7 +43,7 @@ master_doc = 'index' # General information about the project. project = u'GeoPandas' -copyright = u'2013-2014, GeoPandas developers' +copyright = u'2013-2016, GeoPandas developers' # The version info for the project you're documenting, acts as replacement for # |version| and |release|, also used in various other places throughout the From c780783d1cea2982f0c982b69c454166645ccc16 Mon Sep 17 00:00:00 2001 From: Joris Van den Bossche Date: Sat, 11 Jun 2016 00:04:30 +0200 Subject: [PATCH 34/34] DOC: fix readthedocs link --- README.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/README.md b/README.md index a1f04fb..5268f82 100644 --- a/README.md +++ b/README.md @@ -22,7 +22,7 @@ but that may change in the future. Documentation is available at [geopandas.org](http://geopandas.org) (current release) and -[Read the Docs](http://geopandas.readthedocs.io/en/master/) +[Read the Docs](http://geopandas.readthedocs.io/en/latest/) (release and development versions). Install