diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..8340bd6 --- /dev/null +++ b/.gitattributes @@ -0,0 +1 @@ +geopandas/_version.py export-subst diff --git a/CHANGELOG b/CHANGELOG new file mode 100644 index 0000000..58d7c8a --- /dev/null +++ b/CHANGELOG @@ -0,0 +1,37 @@ +Version 0.2.0 +------------- + +Improvements: + +* Complete overhaul of the documentation +* Addition of ``overlay`` to perform spatial overlays with polygons (#142) +* Addition of ``sjoin`` to perform spatial joins (#115, #145, #188) +* Addition of ``__geo_interface__`` that returns a python data structure + to represent the ``GeoSeries`` as a GeoJSON-like ``FeatureCollection`` (#116) + and ``iterfeatures`` method (#178) +* Addition of the ``explode`` (#146) and ``dissolve`` (#310, #311) methods. +* Addition of the ``sindex`` attribute, a Spatial Index using the optional + dependency ``rtree`` (``libspatialindex``) that can be used to speed up + certain operations such as overlays (#140, #141). +* Addition of the ``GeoSeries.ix`` coordinate indexer to slice a GeoSeries based + on a bounding box of the coordinates (#55). +* Improvements to plotting: ability to specify edge colors (#173), support for + the ``vmin``, ``vmax``, ``figsize``, ``linewidth`` keywords (#207), legends + for chloropleth plots (#210), color points by specifying a colormap (#186) or + a single color (#238). +* Larger flexibility of ``to_crs``, accepting both dicts and proj strings (#289) +* Addition of embedded example data, accessible through + ``geopandas.datasets.get_path``. + +API changes: + +* In the ``plot`` method, the ``axes`` keyword is renamed to ``ax`` for + consistency with pandas, and the ``colormap`` keyword is renamed to ``cmap`` + for consistency with matplotlib (#208, #228, #240). + +Bug fixes: + +* Properly handle rows with missing geometries (#139, #193). +* Fix ``GeoSeries.to_json`` (#263). +* Correctly serialize metadata when pickling (#199, #206). +* Fix ``merge`` and ``concat`` to return correct GeoDataFrame (#247, #320, #322). diff --git a/LICENSE.txt b/LICENSE.txt index 15d5aed..7eb8134 100644 --- a/LICENSE.txt +++ b/LICENSE.txt @@ -1,4 +1,4 @@ -Copyright (c) 2013, GeoPandas developers. +Copyright (c) 2013-2016, GeoPandas developers. All rights reserved. Redistribution and use in source and binary forms, with or without diff --git a/MANIFEST.in b/MANIFEST.in new file mode 100644 index 0000000..d7934e2 --- /dev/null +++ b/MANIFEST.in @@ -0,0 +1,2 @@ +include versioneer.py +include geopandas/_version.py diff --git a/README.md b/README.md index 4a410bc..5268f82 100644 --- a/README.md +++ b/README.md @@ -20,6 +20,11 @@ transformed to new coordinate systems with the `to_crs()` method. There is currently no enforcement of like coordinates for operations, but that may change in the future. +Documentation is available at [geopandas.org](http://geopandas.org) +(current release) and +[Read the Docs](http://geopandas.readthedocs.io/en/latest/) +(release and development versions). + Install -------- @@ -113,10 +118,3 @@ GeoPandas also implements alternate constructors that can read any data format r dtype: object ![Convex hulls of New York City boroughs](examples/nyc_hull.png) - -TODO ----- - -- Finish implementing and testing pandas methods on GeoPandas objects -- The current GeoDataFrame does not do very much. -- spatial joins, grouping and more... diff --git a/doc/environment.yml b/doc/environment.yml index 1c81308..0daf4ef 100644 --- a/doc/environment.yml +++ b/doc/environment.yml @@ -7,10 +7,12 @@ dependencies: - shapely - fiona - pyproj +- rtree - six - geopy - matplotlib - descartes +- pysal - sphinx - sphinx_rtd_theme - ipython=4.0.1 diff --git a/doc/source/_static/overlay_operations.png b/doc/source/_static/overlay_operations.png new file mode 100644 index 0000000..38beb8f Binary files /dev/null and b/doc/source/_static/overlay_operations.png differ diff --git a/doc/source/aggregation_with_dissolve.rst b/doc/source/aggregation_with_dissolve.rst new file mode 100644 index 0000000..a191fb5 --- /dev/null +++ b/doc/source/aggregation_with_dissolve.rst @@ -0,0 +1,51 @@ +.. ipython:: python + :suppress: + + import geopandas as gpd + + +Aggregation with dissolve +============================= + +It is often the case that we find ourselves working with spatial data that is more granular than we need. For example, we might have data on sub-national units, but we're actually interested in studying patterns at the level of countries. + +In a non-spatial setting, we aggregate our data using the ``groupby`` function. But when working with spatial data, we need a special tool that can also aggregate geometric features. In the *geopandas* library, that functionality is provided by the ``dissolve`` function. + +``dissolve`` can be thought of as doing three things: (a) it dissolves all the geometries within a given group together into a single geometric feature (using the ``unary_union`` method), and (b) it aggregates all the rows of data in a group using ``groupby.aggregate()``, and (c) it combines those two results. + +``dissolve`` Example +~~~~~~~~~~~~~~~~~~~~~ + +Suppose we are interested in studying continents, but we only have country-level data like the country dataset included in *geopandas*. We can easily convert this to a continent-level dataset. + + +First, let's look at the most simple case where we just want continent shapes and names. By default, ``dissolve`` will pass ``'first'`` to ``groupby.aggregate``. + +.. ipython:: python + + world = gpd.read_file(gpd.datasets.get_path('naturalearth_lowres')) + world = world[['continent', 'geometry']] + continents = world.dissolve(by='continent') + + @savefig continents.png width=5in + continents.plot(); + + continents.head() + +If we are interested in aggregate populations, however, we can pass different functions to the ``dissolve`` method to aggregate populations: + +.. ipython:: python + + world = gpd.read_file(gpd.datasets.get_path('naturalearth_lowres')) + world = world[['continent', 'geometry', 'pop_est']] + continents = world.dissolve(by='continent', aggfunc='sum') + + @savefig continents.png width=5in + continents.plot(column = 'pop_est', scheme='quantiles', cmap='YlOrRd'); + + continents.head() + + + +.. toctree:: + :maxdepth: 2 diff --git a/doc/source/conf.py b/doc/source/conf.py index c713508..d563ab3 100644 --- a/doc/source/conf.py +++ b/doc/source/conf.py @@ -43,18 +43,13 @@ master_doc = 'index' # General information about the project. project = u'GeoPandas' -copyright = u'2013-2014, GeoPandas developers' +copyright = u'2013-2016, GeoPandas developers' # The version info for the project you're documenting, acts as replacement for # |version| and |release|, also used in various other places throughout the # built documents. -d = {} -try: - execfile(os.path.join('..', '..', 'geopandas', 'version.py'), d) - version = release = d['version'] -except: - # FIXME: This shouldn't be hardwired, but should be set one place only - version = release = '0.2.0.dev' +import geopandas +version = release = geopandas.__version__ # The language for content autogenerated by Sphinx. Refer to documentation # for a list of supported languages. diff --git a/doc/source/data_structures.rst b/doc/source/data_structures.rst index 96fb953..c80361d 100644 --- a/doc/source/data_structures.rst +++ b/doc/source/data_structures.rst @@ -4,8 +4,7 @@ :suppress: import geopandas as gpd - world = gpd.GeoDataFrame().from_file('_example_data/naturalearth_lowres.shp') - world = world.rename(columns={'geometry': 'borders'}).set_geometry('borders') + Data Structures ========================================= @@ -70,7 +69,7 @@ Basic Methods Relationship Tests ^^^^^^^^^^^^^^^^^^^ -* ``almost_equals(other)``: is shape almost the same as ``other`` (good when floating point precision issues make shapes slightly different) +* ``geom_almost_equals(other)``: is shape almost the same as ``other`` (good when floating point precision issues make shapes slightly different) * ``contains(other)``: is shape contained within ``other`` * ``intersects(other)``: does shape intersect ``other`` @@ -90,18 +89,27 @@ An example using the ``worlds`` GeoDataFrame: .. ipython:: python + world = gpd.read_file(gpd.datasets.get_path('naturalearth_lowres')) + world.head() #Plot countries @savefig world_borders.png width=3in world.plot(); -Currently, the column named "borders" with country borders is the active +Currently, the column named "geometry" with country borders is the active geometry column: .. ipython:: python world.geometry.name +We can also rename this column to "borders": + +.. ipython:: python + + world = world.rename(columns={'geometry': 'borders'}).set_geometry('borders') + world.geometry.name + Now, we create centroids and make it the geometry: .. ipython:: python diff --git a/doc/source/geometric_manipulations.rst b/doc/source/geometric_manipulations.rst index 2996ee2..b1b256c 100644 --- a/doc/source/geometric_manipulations.rst +++ b/doc/source/geometric_manipulations.rst @@ -1,92 +1,73 @@ Geometric Manipulations ======================== +*geopandas* makes available all the tools for geometric manipulations in the `*shapely* library `_. - - -Set-theoretic Methods -~~~~~~~~~~~~~~~~~~~~~ - -.. attribute:: GeoSeries.boundary - - Returns a ``GeoSeries`` of lower dimensional objects representing - each geometries's set-theoretic `boundary`. - -.. method:: GeoSeries.difference(other) - - Returns a ``GeoSeries`` of the points in each geometry that - are not in the *other* object. - -.. method:: GeoSeries.intersection(other) - - Returns a ``GeoSeries`` of the intersection of each object with the `other` - geometric object. - -.. method:: GeoSeries.symmetric_difference(other) - - Returns a ``GeoSeries`` of the points in each object not in the `other` - geometric object, and the points in the `other` not in this object. - -.. method:: GeoSeries.union(other) - - Returns a ``GeoSeries`` of the union of points from each object and the - `other` geometric object. - - -.. attribute:: GeoSeries.unary_union - - Return a geometry containing the union of all geometries in the ``GeoSeries``. - +Note that documentation for all set-theoretic tools for creating new shapes using the relationship between two different spatial datasets -- like creating intersections, or differences -- can be found on the :doc:`set operations ` page. Constructive Methods -~~~~~~~~~~~~~~~~~~~~~ +~~~~~~~~~~~~~~~~~~~~ -.. method:: GeoSeries.buffer(distance, resolution=16) +.. method:: GeoSeries.buffer(distance, resolution=16) Returns a ``GeoSeries`` of geometries representing all points within a given `distance` of each geometric object. -.. attribute:: GeoSeries.convex_hull +.. attribute:: GeoSeries.boundary + + Returns a ``GeoSeries`` of lower dimensional objects representing + each geometries's set-theoretic `boundary`. + +.. attribute:: GeoSeries.centroid + + Returns a ``GeoSeries`` of points for each geometric centroid. + +.. attribute:: GeoSeries.convex_hull Returns a ``GeoSeries`` of geometries representing the smallest convex `Polygon` containing all the points in each object unless the number of points in the object is less than three. For two points, the convex hull collapses to a `LineString`; for 1, a `Point`. -.. attribute:: GeoSeries.envelope +.. attribute:: GeoSeries.envelope Returns a ``GeoSeries`` of geometries representing the point or smallest rectangular polygon (with sides parallel to the coordinate axes) that contains each object. -.. method:: GeoSeries.simplify(tolerance, preserve_topology=True) +.. method:: GeoSeries.simplify(tolerance, preserve_topology=True) Returns a ``GeoSeries`` containing a simplified representation of each object. -Affine transformations -~~~~~~~~~~~~~~~~~~~~~~~ +.. attribute:: GeoSeries.unary_union -.. method:: GeoSeries.rotate(self, angle, origin='center', use_radians=False) + Return a geometry containing the union of all geometries in the ``GeoSeries``. + + +Affine transformations +~~~~~~~~~~~~~~~~~~~~~~~~ + +.. method:: GeoSeries.rotate(self, angle, origin='center', use_radians=False) Rotate the coordinates of the GeoSeries. -.. method:: GeoSeries.scale(self, xfact=1.0, yfact=1.0, zfact=1.0, origin='center') +.. method:: GeoSeries.scale(self, xfact=1.0, yfact=1.0, zfact=1.0, origin='center') Scale the geometries of the GeoSeries along each (x, y, z) dimensio. -.. method:: GeoSeries.skew(self, angle, origin='center', use_radians=False) +.. method:: GeoSeries.skew(self, angle, origin='center', use_radians=False) Shear/Skew the geometries of the GeoSeries by angles along x and y dimensions. -.. method:: GeoSeries.translate(self, angle, origin='center', use_radians=False) +.. method:: GeoSeries.translate(self, angle, origin='center', use_radians=False) Shift the coordinates of the GeoSeries. -`Aggregating methods` - +Examples of Geometric Manipulations +------------------------------------ .. sourcecode:: python @@ -146,7 +127,7 @@ GeoPandas also implements alternate constructors that can read any data format r 3 Brooklyn 1.959432e+09 726568.946340 4 Queens 3.049947e+09 861038.479299 5 Staten Island 1.623853e+09 330385.036974 - + geometry BoroCode 1 (POLYGON ((981219.0557861328125000 188655.3157... @@ -156,7 +137,7 @@ GeoPandas also implements alternate constructors that can read any data format r 5 (POLYGON ((970217.0223999023437500 145643.3322... .. image:: _static/nyc.png - + .. sourcecode:: python >>> boros['geometry'].convex_hull @@ -201,7 +182,7 @@ just use: >>> holes = boros['geometry'].intersection(mp) .. image:: _static/holes.png - + and to get the area outside of the holes: .. sourcecode:: python @@ -209,7 +190,7 @@ and to get the area outside of the holes: >>> boros_with_holes = boros['geometry'].difference(mp) .. image:: _static/boros_with_holes.png - + Note that this can be simplified a bit, since ``geometry`` is available as an attribute on a ``GeoDataFrame``, and the ``intersection`` and ``difference`` methods are implemented with the @@ -239,5 +220,3 @@ borough that are in the holes: .. toctree:: :maxdepth: 2 - - diff --git a/doc/source/index.rst b/doc/source/index.rst index 17d8c81..0395b88 100644 --- a/doc/source/index.rst +++ b/doc/source/index.rst @@ -32,6 +32,8 @@ such as PostGIS. Making Maps Managing Projections Geometric Manipulations + Set Operations with overlay + Aggregation with dissolve Merging Data Geocoding Reference to All Attributes and Methods diff --git a/doc/source/install.rst b/doc/source/install.rst index 4153f5a..37d80e7 100644 --- a/doc/source/install.rst +++ b/doc/source/install.rst @@ -20,9 +20,7 @@ You may install the latest development version by cloning the pip install . It is also possible to install the latest development version -available on PyPI with `pip` by adding the ``--pre`` flag for pip 1.4 -and later, or to use `pip` to install directly from the GitHub -repository with:: +directly from the GitHub repository with:: pip install git+git://github.com/geopandas/geopandas.git @@ -32,7 +30,7 @@ Dependencies Installation via `conda` should also install all dependencies, but a complete list is as follows: - `numpy`_ -- `pandas`_ (version 0.13 or later) +- `pandas`_ (version 0.15.2 or later) - `shapely`_ - `fiona`_ - `six`_ diff --git a/doc/source/mapping.rst b/doc/source/mapping.rst index 2ce1cb6..ff75960 100644 --- a/doc/source/mapping.rst +++ b/doc/source/mapping.rst @@ -4,19 +4,25 @@ :suppress: import geopandas as gpd - world = gpd.GeoDataFrame().from_file('_example_data/naturalearth_lowres.shp') - cities = gpd.GeoDataFrame().from_file('_example_data/naturalearth_cities.shp') - Mapping Tools ========================================= -*geopandas* provides a high-level interface to the ``matplotlib`` library for making maps. Mapping shapes is as easy as using the ``plot()`` method on a ``GeoSeries`` or ``GeoDataFrame``. +*geopandas* provides a high-level interface to the ``matplotlib`` library for making maps. Mapping shapes is as easy as using the ``plot()`` method on a ``GeoSeries`` or ``GeoDataFrame``. + +Loading some example data: .. ipython:: python - + + world = gpd.read_file(gpd.datasets.get_path('naturalearth_lowres')) + cities = gpd.read_file(gpd.datasets.get_path('naturalearth_cities')) + +We can now plot those GeoDataFrames: + +.. ipython:: python + # Examine country GeoDataFrame world.head() @@ -24,13 +30,13 @@ Mapping Tools @savefig world_randomcolors.png width=5in world.plot(); -Note that in general, any options one can pass to `pyplot `_ in ``matplotlib`` (or `style options that work for lines `_) can be passed to the ``plot()`` method. +Note that in general, any options one can pass to `pyplot `_ in ``matplotlib`` (or `style options that work for lines `_) can be passed to the ``plot()`` method. Chloropleth Maps ----------------- -*geopandas* makes it easy to create Chloropleth maps (maps where the color of each shape is based on the value of an associated variable). Simply use the plot command with the ``column`` argument set to the column whose values you want used to assign colors. +*geopandas* makes it easy to create Chloropleth maps (maps where the color of each shape is based on the value of an associated variable). Simply use the plot command with the ``column`` argument set to the column whose values you want used to assign colors. .. ipython:: python @@ -52,7 +58,7 @@ One can also modify the colors used by ``plot`` with the ``cmap`` option (for a world.plot(column='gdp_per_cap', cmap='OrRd'); -The way color maps are scaled can also be manipulated with the ``scheme`` option (if you have ``pysal`` installed, which can be accomplished via ``conda install pysal``). By default, ``scheme`` is set to 'equal_intervals', but it can also be adjusted to any other `pysal option `_, like 'quantiles', 'percentiles', etc. +The way color maps are scaled can also be manipulated with the ``scheme`` option (if you have ``pysal`` installed, which can be accomplished via ``conda install pysal``). By default, ``scheme`` is set to 'equal_intervals', but it can also be adjusted to any other `pysal option `_, like 'quantiles', 'percentiles', etc. .. ipython:: python @@ -63,14 +69,14 @@ The way color maps are scaled can also be manipulated with the ``scheme`` option Maps with Layers ----------------- -There are two strategies for making a map with multiple layers -- one more succinct, and one that is a littel more flexible. +There are two strategies for making a map with multiple layers -- one more succinct, and one that is a littel more flexible. -Before combining maps, however, remember to always ensure they share a common CRS (so they will align). +Before combining maps, however, remember to always ensure they share a common CRS (so they will align). .. ipython:: python - + # Look at capitals - # Note use of standard `pyplot` line style options + # Note use of standard `pyplot` line style options @savefig capitals.png width=5in cities.plot(marker='*', color='green', markersize=5); @@ -96,10 +102,10 @@ Before combining maps, however, remember to always ensure they share a common CR import matplotlib.pyplot as plt fig, ax = plt.subplots() - # set aspect to equal. This is done automatically - # when using *geopandas* plot on it's own, but not when - # working with pyplot directly. - ax.set_aspect('equal') + # set aspect to equal. This is done automatically + # when using *geopandas* plot on it's own, but not when + # working with pyplot directly. + ax.set_aspect('equal') world.plot(ax=ax, color='white') cities.plot(ax=ax, marker='o', color='red', markersize=5) @@ -112,4 +118,3 @@ Other Resources Links to jupyter Notebooks for different mapping tasks: `Making Heat Maps `_ - diff --git a/doc/source/mergingdata.rst b/doc/source/mergingdata.rst index 66233af..3ebd69c 100644 --- a/doc/source/mergingdata.rst +++ b/doc/source/mergingdata.rst @@ -1,15 +1,77 @@ +.. currentmodule:: geopandas + +.. ipython:: python + :suppress: + + import geopandas as gpd + Merging Data ========================================= +There are two ways to combine datasets in *geopandas* -- attribute joins and spatial joins. + +In an attribute join, a ``GeoSeries`` or ``GeoDataFrame`` is combined with a regular *pandas* ``Series`` or ``DataFrame`` based on a common variable. This is analogous to normal merging or joining in *pandas*. + +In a Spatial Join, observations from to ``GeoSeries`` or ``GeoDataFrames`` are combined based on their spatial relationship to one another. + +In the following examples, we use these datasets: + +.. ipython:: python + + world = gpd.read_file(gpd.datasets.get_path('naturalearth_lowres')) + cities = gpd.read_file(gpd.datasets.get_path('naturalearth_cities')) + + # For attribute join + country_shapes = world[['geometry', 'iso_a3']] + country_names = world[['name', 'iso_a3']] + + # For spatial join + countries = world[['geometry', 'name']] + countries = countries.rename(columns={'name':'country'}) + Attribute Joins ---------------- -[TO BE COMPLETED -- EXAMPLES OF JOINING GDF WITH PANDAS DATAFRAME] +Attribute joins are accomplished using the ``merge`` method. In general, it is recommended to use the ``merge`` method called from the spatial dataset. With that said, the stand-alone ``merge`` function will work if the GeoDataFrame is in the ``left`` argument; if a DataFrame is in the ``left`` argument and a GeoDataFrame is in the ``right`` position, the result will no longer be a GeoDataFrame. + + +For example, consider the following merge that adds full names to a ``GeoDataFrame`` that initially has only ISO codes for each country by merging it with a *pandas* ``DataFrame``. + +.. ipython:: python + + # `country_shapes` is GeoDataFrame with country shapes and iso codes + country_shapes.head() + + # `country_names` is DataFrame with country names and iso codes + country_names.head() + + # Merge with `merge` method on shared variable (iso codes): + country_shapes = country_shapes.merge(country_names, on='iso_a3') + country_shapes.head() + Spatial Joins ---------------- -[TO BE COMPLETED -- EXAMPLES OF SPATIAL JOINS] +In a Spatial Join, two geometry objects are merged based on their spatial relationship to one another. + +.. ipython:: python + + + # One GeoDataFrame of countries, one of Cities. + # Want to merge so we can get each city's country. + countries.head() + cities.head() + + # Execute spatial join + + cities_with_country = gpd.sjoin(cities, countries, how="inner", op='intersects') + cities_with_country.head() + + +The ``op`` options determines the type of join operation to apply. ``op`` can be set to "intersects", "within" or "contains" (these are all equivalent when joining points to polygons, but differ when joining polygons to other polygons or lines). + +Note more complicated spatial relationships can be studied by combining geometric operations with spatial join. To find all polygons within a given distance of a point, for example, one can first use the ``buffer`` method to expand each point into a circle of appropriate radius, then intersect those buffered circles with the polygons in question. diff --git a/doc/source/projections.rst b/doc/source/projections.rst index de4f03c..bfca83c 100644 --- a/doc/source/projections.rst +++ b/doc/source/projections.rst @@ -4,8 +4,6 @@ :suppress: import geopandas as gpd - world = gpd.GeoDataFrame().from_file('_example_data/naturalearth_lowres.shp') - Managing Projections @@ -18,11 +16,11 @@ Coordinate Reference Systems CRS are important because the geometric shapes in a GeoSeries or GeoDataFrame object are simply a collection of coordinates in an arbitrary space. A CRS tells Python how those coordinates related to places on the Earth. -CRS are referred to using codes called `proj4 strings `_. You can find the codes for most commonly used projections from `www.spatialreference.org `_ or `remotesensing.org `_. +CRS are referred to using codes called `proj4 strings `_. You can find the codes for most commonly used projections from `www.spatialreference.org `_ or `remotesensing.org `_. -The same CRS can often be referred to in many ways. For example, one of the most commonly used CRS is the WGS84 latitude-longitude projection. One `proj4` representation of this projection is: ``"+proj=longlat +ellps=WGS84 +datum=WGS84 +no_defs"``. But common projections can also be referred to by `EPSG` codes, so this same projection can also called using the `proj4` string ``"+init=epsg:4326"``. +The same CRS can often be referred to in many ways. For example, one of the most commonly used CRS is the WGS84 latitude-longitude projection. One `proj4` representation of this projection is: ``"+proj=longlat +ellps=WGS84 +datum=WGS84 +no_defs"``. But common projections can also be referred to by `EPSG` codes, so this same projection can also called using the `proj4` string ``"+init=epsg:4326"``. -*geopandas* can accept lots of representations of CRS, including the `proj4` string itself (``"+proj=longlat +ellps=WGS84 +datum=WGS84 +no_defs"``) or parameters broken out in a dictionary: ``{'proj': 'latlong', 'ellps': 'WGS84', 'datum': 'WGS84', 'no_defs': True}``). In addition, some functions will take `EPSG` codes directly. +*geopandas* can accept lots of representations of CRS, including the `proj4` string itself (``"+proj=longlat +ellps=WGS84 +datum=WGS84 +no_defs"``) or parameters broken out in a dictionary: ``{'proj': 'latlong', 'ellps': 'WGS84', 'datum': 'WGS84', 'no_defs': True}``). In addition, some functions will take `EPSG` codes directly. For reference, a few very common projections and their proj4 strings: @@ -33,18 +31,18 @@ For reference, a few very common projections and their proj4 strings: Setting a Projection ---------------------- -There are two relevant operations for projections: setting a projection and re-projecting. +There are two relevant operations for projections: setting a projection and re-projecting. Setting a projection may be necessary when for some reason *geopandas* has coordinate data (x-y values), but no information about how those coordinates refer to locations in the real world. Setting a projection is how one tells *geopandas* how to interpret coordinates. If no CRS is set, *geopandas* geometry operations will still work, but coordinate transformations will not be possible and exported files may not be interpreted correctly by other software. -Be aware that **most of the time** you don't have to set a projection. Data loaded from a reputable source (using the ``from_file()`` command) *should* always include projection information. You can see an objects current CRS through the ``crs`` attribute: ``my_geoseries.crs``. +Be aware that **most of the time** you don't have to set a projection. Data loaded from a reputable source (using the ``from_file()`` command) *should* always include projection information. You can see an objects current CRS through the ``crs`` attribute: ``my_geoseries.crs``. From time to time, however, you may get data that does not include a projection. In this situation, you have to set the CRS so *geopandas* knows how to interpret the coordinates. For example, if you convert a spreadsheet of latitudes and longitudes into a GeoSeries by hand, you would set the projection by assigning the WGS84 latitude-longitude CRS to the ``crs`` attribute: .. sourcecode:: python - + my_geoseries.crs = {'init' :'epsg:4326'} @@ -54,7 +52,10 @@ Re-Projecting Re-projecting is the process of changing the representation of locations from one coordinate system to another. All projections of locations on the Earth into a two-dimensional plane `are distortions `_, the projection that is best for your application may be different from the projection associated with the data you import. In these cases, data can be re-projected using the ``to_crs`` command: .. ipython:: python - + + # load example data + world = gpd.read_file(gpd.datasets.get_path('naturalearth_lowres')) + # Check original projection # (it's Platte Carre! x-y are long and lat) world.crs @@ -68,4 +69,3 @@ Re-projecting is the process of changing the representation of locations from on world = world.to_crs({'init': 'epsg:3395'}) # world.to_crs(epsg=3395) would also work @savefig world_reproj.png width=3in world.plot(); - diff --git a/doc/source/reference.rst b/doc/source/reference.rst index a05e840..2b95ab7 100644 --- a/doc/source/reference.rst +++ b/doc/source/reference.rst @@ -124,15 +124,6 @@ The following Shapely methods and attributes are available on `Set-theoretic Methods` -.. attribute:: GeoSeries.boundary - - Returns a ``GeoSeries`` of lower dimensional objects representing - each geometries's set-theoretic `boundary`. - -.. attribute:: GeoSeries.centroid - - Returns a ``GeoSeries`` of points for each geometric centroid. - .. method:: GeoSeries.difference(other) Returns a ``GeoSeries`` of the points in each geometry that @@ -160,6 +151,15 @@ The following Shapely methods and attributes are available on Returns a ``GeoSeries`` of geometries representing all points within a given `distance` of each geometric object. +.. attribute:: GeoSeries.boundary + + Returns a ``GeoSeries`` of lower dimensional objects representing + each geometries's set-theoretic `boundary`. + +.. attribute:: GeoSeries.centroid + + Returns a ``GeoSeries`` of points for each geometric centroid. + .. attribute:: GeoSeries.convex_hull Returns a ``GeoSeries`` of geometries representing the smallest diff --git a/doc/source/set_operations.rst b/doc/source/set_operations.rst new file mode 100644 index 0000000..2db553b --- /dev/null +++ b/doc/source/set_operations.rst @@ -0,0 +1,195 @@ +.. ipython:: python + :suppress: + + import geopandas as gpd + + +Set-Operations with Overlay +============================ + +When working with multiple spatial datasets -- especially multiple *polygon* or +*line* datasets -- users often wish to create new shapes based on places where +those datasets overlap (or don't overlap). These manipulations are often +referred using the language of sets -- intersections, unions, and differences. +These types of operations are made available in the *geopandas* library through +the ``overlay`` function. + +The basic idea is demonstrated by the graphic below but keep in mind that +overlays operate at the DataFrame level, not on individual geometries, and the +properties from both are retained. In effect, for every shape in the first +GeoDataFrame, this operation is executed against every other shape in the other +GeoDataFrame: + +.. image:: _static/overlay_operations.png + +**Source: QGIS Documentation** + +(Note to users familiar with the *shapely* library: ``overlay`` can be thought +of as offering versions of the standard *shapely* set-operations that deal with +the complexities of applying set operations to two *GeoSeries*. The standard +*shapely* set-operations are also available as ``GeoSeries`` methods.) + + +The different Overlay operations +-------------------------------- + +First, we create some example data: + +.. ipython:: python + + from shapely.geometry import Polygon + polys1 = gpd.GeoSeries([Polygon([(0,0), (2,0), (2,2), (0,2)]), + Polygon([(2,2), (4,2), (4,4), (2,4)])]) + polys2 = gpd.GeoSeries([Polygon([(1,1), (3,1), (3,3), (1,3)]), + Polygon([(3,3), (5,3), (5,5), (3,5)])]) + + df1 = gpd.GeoDataFrame({'geometry': polys1, 'df1':[1,2]}) + df2 = gpd.GeoDataFrame({'geometry': polys2, 'df2':[1,2]}) + +These two GeoDataFrames have some overlapping areas: + +.. ipython:: python + + ax = df1.plot(color='red'); + @savefig overlay_example.png width=5in + df2.plot(ax=ax, color='green'); + +We illustrate the different overlay modes with the above example. +The ``overlay`` function will determine the set of all individual geometries +from overlaying the two input GeoDataFrames. This result covers the area covered +by the two input GeoDataFrames, and also preserves all unique regions defined by +the combined boundaries of the two GeoDataFrames. + +When using ``how='union'``, all those possible geometries are returned: + +.. ipython:: python + + res_union = gpd.overlay(df1, df2, how='union') + res_union + + ax = res_union.plot() + df1.plot(ax=ax, facecolor='none'); + @savefig overlay_example_union.png width=5in + df2.plot(ax=ax, facecolor='none'); + +The other ``how`` operations will return different subsets of those geometries. +With ``how='intersection'``, it returns only those geometries that are contained +by both GeoDataFrames: + +.. ipython:: python + + res_intersection = gpd.overlay(df1, df2, how='intersection') + res_intersection + + ax = res_intersection.plot() + df1.plot(ax=ax, facecolor='none'); + @savefig overlay_example_intersection.png width=5in + df2.plot(ax=ax, facecolor='none'); + +``how='symmetric_difference'`` is the opposite of ``'intersection'`` and returns +the geometries that are only part of one of the GeoDataFrames but not of both: + +.. ipython:: python + + res_symdiff = gpd.overlay(df1, df2, how='symmetric_difference') + res_symdiff + + ax = res_symdiff.plot() + df1.plot(ax=ax, facecolor='none'); + @savefig overlay_example_symdiff.png width=5in + df2.plot(ax=ax, facecolor='none'); + +To obtain the geometries that are part of ``df1`` but are not contained in +``df2``, you can use ``how='difference'``: + +.. ipython:: python + + res_difference = gpd.overlay(df1, df2, how='difference') + res_difference + + ax = res_difference.plot() + df1.plot(ax=ax, facecolor='none'); + @savefig overlay_example_difference.png width=5in + df2.plot(ax=ax, facecolor='none'); + +Finally, with ``how='identity'``, the result consists of the surface of ``df1``, +but with the geometries obtained from overlaying ``df1`` with ``df2``: + +.. ipython:: python + + res_identity = gpd.overlay(df1, df2, how='identity') + res_identity + + ax = res_identity.plot() + df1.plot(ax=ax, facecolor='none'); + @savefig overlay_example_identity.png width=5in + df2.plot(ax=ax, facecolor='none'); + + +Overlay Countries Example +------------------------- + +First, we load the countries and cities example datasets and select : + +.. ipython:: python + + world = gpd.read_file(gpd.datasets.get_path('naturalearth_lowres')) + capitals = gpd.read_file(gpd.datasets.get_path('naturalearth_cities')) + + # Select South Amarica and some columns + countries = world[world['continent'] == "South America"] + countries = countries[['geometry', 'name']] + + # Project to crs that uses meters as distance measure + countries = countries.to_crs('+init=epsg:3395') + capitals = capitals.to_crs('+init=epsg:3395') + +To illustrate the ``overlay`` function, consider the following case in which one +wishes to identify the "core" portion of each country -- defined as areas within +500km of a capital -- using a ``GeoDataFrame`` of countries and a +``GeoDataFrame`` of capitals. + +.. ipython:: python + + # Look at countries: + @savefig world_basic.png width=5in + countries.plot(); + + # Now buffer cities to find area within 500km. + # Check CRS -- World Mercator, units of meters. + capitals.crs + + # make 500km buffer + capitals['geometry']= capitals.buffer(500000) + @savefig capital_buffers.png width=5in + capitals.plot(); + + +To select only the portion of countries within 500km of a capital, we specify the ``how`` option to be "intersect", which creates a new set of polygons where these two layers overlap: + +.. ipython:: python + + country_cores = gpd.overlay(countries, capitals, how='intersection') + @savefig country_cores.png width=5in + country_cores.plot(); + +Changing the "how" option allows for different types of overlay operations. For example, if we were interested in the portions of countries *far* from capitals (the peripheries), we would compute the difference of the two. + +.. ipython:: python + + country_peripheries = gpd.overlay(countries, capitals, how='difference') + @savefig country_peripheries.png width=5in + country_peripheries.plot(); + + + + +More Examples +------------- + +A larger set of examples of the use of ``overlay`` can be found `here `_ + + + +.. toctree:: + :maxdepth: 2 diff --git a/geopandas/__init__.py b/geopandas/__init__.py index 2780f17..77eb04b 100644 --- a/geopandas/__init__.py +++ b/geopandas/__init__.py @@ -1,16 +1,19 @@ -try: - from geopandas.version import version as __version__ -except ImportError: - __version__ = '0.2.0.dev-unknown' - from geopandas.geoseries import GeoSeries from geopandas.geodataframe import GeoDataFrame from geopandas.io.file import read_file from geopandas.io.sql import read_postgis +from geopandas.tools import sjoin +from geopandas.tools import overlay + +import geopandas.datasets # make the interactive namespace easier to use # for `from geopandas import *` demos. import geopandas as gpd import pandas as pd import numpy as np + +from ._version import get_versions +__version__ = get_versions()['version'] +del get_versions diff --git a/geopandas/_version.py b/geopandas/_version.py new file mode 100644 index 0000000..4604e70 --- /dev/null +++ b/geopandas/_version.py @@ -0,0 +1,484 @@ + +# This file helps to compute a version number in source trees obtained from +# git-archive tarball (such as those provided by githubs download-from-tag +# feature). Distribution tarballs (built by setup.py sdist) and build +# directories (produced by setup.py build) will contain a much shorter file +# that just contains the computed version number. + +# This file is released into the public domain. Generated by +# versioneer-0.16 (https://github.com/warner/python-versioneer) + +"""Git implementation of _version.py.""" + +import errno +import os +import re +import subprocess +import sys + + +def get_keywords(): + """Get the keywords needed to look up the version information.""" + # these strings will be replaced by git during git-archive. + # setup.py/versioneer.py will grep for the variable names, so they must + # each be defined on a line of their own. _version.py will just call + # get_keywords(). + git_refnames = "$Format:%d$" + git_full = "$Format:%H$" + keywords = {"refnames": git_refnames, "full": git_full} + return keywords + + +class VersioneerConfig: + """Container for Versioneer configuration parameters.""" + + +def get_config(): + """Create, populate and return the VersioneerConfig() object.""" + # these strings are filled in when 'setup.py versioneer' creates + # _version.py + cfg = VersioneerConfig() + cfg.VCS = "git" + cfg.style = "pep440" + cfg.tag_prefix = "v" + cfg.parentdir_prefix = "geopandas-" + cfg.versionfile_source = "geopandas/_version.py" + cfg.verbose = False + return cfg + + +class NotThisMethod(Exception): + """Exception raised if a method is not valid for the current scenario.""" + + +LONG_VERSION_PY = {} +HANDLERS = {} + + +def register_vcs_handler(vcs, method): # decorator + """Decorator to mark a method as the handler for a particular VCS.""" + def decorate(f): + """Store f in HANDLERS[vcs][method].""" + if vcs not in HANDLERS: + HANDLERS[vcs] = {} + HANDLERS[vcs][method] = f + return f + return decorate + + +def run_command(commands, args, cwd=None, verbose=False, hide_stderr=False): + """Call the given command(s).""" + assert isinstance(commands, list) + p = None + for c in commands: + try: + dispcmd = str([c] + args) + # remember shell=False, so use git.cmd on windows, not just git + p = subprocess.Popen([c] + args, cwd=cwd, stdout=subprocess.PIPE, + stderr=(subprocess.PIPE if hide_stderr + else None)) + break + except EnvironmentError: + e = sys.exc_info()[1] + if e.errno == errno.ENOENT: + continue + if verbose: + print("unable to run %s" % dispcmd) + print(e) + return None + else: + if verbose: + print("unable to find command, tried %s" % (commands,)) + return None + stdout = p.communicate()[0].strip() + if sys.version_info[0] >= 3: + stdout = stdout.decode() + if p.returncode != 0: + if verbose: + print("unable to run %s (error)" % dispcmd) + return None + return stdout + + +def versions_from_parentdir(parentdir_prefix, root, verbose): + """Try to determine the version from the parent directory name. + + Source tarballs conventionally unpack into a directory that includes + both the project name and a version string. + """ + dirname = os.path.basename(root) + if not dirname.startswith(parentdir_prefix): + if verbose: + print("guessing rootdir is '%s', but '%s' doesn't start with " + "prefix '%s'" % (root, dirname, parentdir_prefix)) + raise NotThisMethod("rootdir doesn't start with parentdir_prefix") + return {"version": dirname[len(parentdir_prefix):], + "full-revisionid": None, + "dirty": False, "error": None} + + +@register_vcs_handler("git", "get_keywords") +def git_get_keywords(versionfile_abs): + """Extract version information from the given file.""" + # the code embedded in _version.py can just fetch the value of these + # keywords. When used from setup.py, we don't want to import _version.py, + # so we do it with a regexp instead. This function is not used from + # _version.py. + keywords = {} + try: + f = open(versionfile_abs, "r") + for line in f.readlines(): + if line.strip().startswith("git_refnames ="): + mo = re.search(r'=\s*"(.*)"', line) + if mo: + keywords["refnames"] = mo.group(1) + if line.strip().startswith("git_full ="): + mo = re.search(r'=\s*"(.*)"', line) + if mo: + keywords["full"] = mo.group(1) + f.close() + except EnvironmentError: + pass + return keywords + + +@register_vcs_handler("git", "keywords") +def git_versions_from_keywords(keywords, tag_prefix, verbose): + """Get version information from git keywords.""" + if not keywords: + raise NotThisMethod("no keywords at all, weird") + refnames = keywords["refnames"].strip() + if refnames.startswith("$Format"): + if verbose: + print("keywords are unexpanded, not using") + raise NotThisMethod("unexpanded keywords, not a git-archive tarball") + refs = set([r.strip() for r in refnames.strip("()").split(",")]) + # starting in git-1.8.3, tags are listed as "tag: foo-1.0" instead of + # just "foo-1.0". If we see a "tag: " prefix, prefer those. + TAG = "tag: " + tags = set([r[len(TAG):] for r in refs if r.startswith(TAG)]) + if not tags: + # Either we're using git < 1.8.3, or there really are no tags. We use + # a heuristic: assume all version tags have a digit. The old git %d + # expansion behaves like git log --decorate=short and strips out the + # refs/heads/ and refs/tags/ prefixes that would let us distinguish + # between branches and tags. By ignoring refnames without digits, we + # filter out many common branch names like "release" and + # "stabilization", as well as "HEAD" and "master". + tags = set([r for r in refs if re.search(r'\d', r)]) + if verbose: + print("discarding '%s', no digits" % ",".join(refs-tags)) + if verbose: + print("likely tags: %s" % ",".join(sorted(tags))) + for ref in sorted(tags): + # sorting will prefer e.g. "2.0" over "2.0rc1" + if ref.startswith(tag_prefix): + r = ref[len(tag_prefix):] + if verbose: + print("picking %s" % r) + return {"version": r, + "full-revisionid": keywords["full"].strip(), + "dirty": False, "error": None + } + # no suitable tags, so version is "0+unknown", but full hex is still there + if verbose: + print("no suitable tags, using unknown + full revision id") + return {"version": "0+unknown", + "full-revisionid": keywords["full"].strip(), + "dirty": False, "error": "no suitable tags"} + + +@register_vcs_handler("git", "pieces_from_vcs") +def git_pieces_from_vcs(tag_prefix, root, verbose, run_command=run_command): + """Get version from 'git describe' in the root of the source tree. + + This only gets called if the git-archive 'subst' keywords were *not* + expanded, and _version.py hasn't already been rewritten with a short + version string, meaning we're inside a checked out source tree. + """ + if not os.path.exists(os.path.join(root, ".git")): + if verbose: + print("no .git in %s" % root) + raise NotThisMethod("no .git directory") + + GITS = ["git"] + if sys.platform == "win32": + GITS = ["git.cmd", "git.exe"] + # if there is a tag matching tag_prefix, this yields TAG-NUM-gHEX[-dirty] + # if there isn't one, this yields HEX[-dirty] (no NUM) + describe_out = run_command(GITS, ["describe", "--tags", "--dirty", + "--always", "--long", + "--match", "%s*" % tag_prefix], + cwd=root) + # --long was added in git-1.5.5 + if describe_out is None: + raise NotThisMethod("'git describe' failed") + describe_out = describe_out.strip() + full_out = run_command(GITS, ["rev-parse", "HEAD"], cwd=root) + if full_out is None: + raise NotThisMethod("'git rev-parse' failed") + full_out = full_out.strip() + + pieces = {} + pieces["long"] = full_out + pieces["short"] = full_out[:7] # maybe improved later + pieces["error"] = None + + # parse describe_out. It will be like TAG-NUM-gHEX[-dirty] or HEX[-dirty] + # TAG might have hyphens. + git_describe = describe_out + + # look for -dirty suffix + dirty = git_describe.endswith("-dirty") + pieces["dirty"] = dirty + if dirty: + git_describe = git_describe[:git_describe.rindex("-dirty")] + + # now we have TAG-NUM-gHEX or HEX + + if "-" in git_describe: + # TAG-NUM-gHEX + mo = re.search(r'^(.+)-(\d+)-g([0-9a-f]+)$', git_describe) + if not mo: + # unparseable. Maybe git-describe is misbehaving? + pieces["error"] = ("unable to parse git-describe output: '%s'" + % describe_out) + return pieces + + # tag + full_tag = mo.group(1) + if not full_tag.startswith(tag_prefix): + if verbose: + fmt = "tag '%s' doesn't start with prefix '%s'" + print(fmt % (full_tag, tag_prefix)) + pieces["error"] = ("tag '%s' doesn't start with prefix '%s'" + % (full_tag, tag_prefix)) + return pieces + pieces["closest-tag"] = full_tag[len(tag_prefix):] + + # distance: number of commits since tag + pieces["distance"] = int(mo.group(2)) + + # commit: short hex revision ID + pieces["short"] = mo.group(3) + + else: + # HEX: no tags + pieces["closest-tag"] = None + count_out = run_command(GITS, ["rev-list", "HEAD", "--count"], + cwd=root) + pieces["distance"] = int(count_out) # total number of commits + + return pieces + + +def plus_or_dot(pieces): + """Return a + if we don't already have one, else return a .""" + if "+" in pieces.get("closest-tag", ""): + return "." + return "+" + + +def render_pep440(pieces): + """Build up version string, with post-release "local version identifier". + + Our goal: TAG[+DISTANCE.gHEX[.dirty]] . Note that if you + get a tagged build and then dirty it, you'll get TAG+0.gHEX.dirty + + Exceptions: + 1: no tags. git_describe was just HEX. 0+untagged.DISTANCE.gHEX[.dirty] + """ + if pieces["closest-tag"]: + rendered = pieces["closest-tag"] + if pieces["distance"] or pieces["dirty"]: + rendered += plus_or_dot(pieces) + rendered += "%d.g%s" % (pieces["distance"], pieces["short"]) + if pieces["dirty"]: + rendered += ".dirty" + else: + # exception #1 + rendered = "0+untagged.%d.g%s" % (pieces["distance"], + pieces["short"]) + if pieces["dirty"]: + rendered += ".dirty" + return rendered + + +def render_pep440_pre(pieces): + """TAG[.post.devDISTANCE] -- No -dirty. + + Exceptions: + 1: no tags. 0.post.devDISTANCE + """ + if pieces["closest-tag"]: + rendered = pieces["closest-tag"] + if pieces["distance"]: + rendered += ".post.dev%d" % pieces["distance"] + else: + # exception #1 + rendered = "0.post.dev%d" % pieces["distance"] + return rendered + + +def render_pep440_post(pieces): + """TAG[.postDISTANCE[.dev0]+gHEX] . + + The ".dev0" means dirty. Note that .dev0 sorts backwards + (a dirty tree will appear "older" than the corresponding clean one), + but you shouldn't be releasing software with -dirty anyways. + + Exceptions: + 1: no tags. 0.postDISTANCE[.dev0] + """ + if pieces["closest-tag"]: + rendered = pieces["closest-tag"] + if pieces["distance"] or pieces["dirty"]: + rendered += ".post%d" % pieces["distance"] + if pieces["dirty"]: + rendered += ".dev0" + rendered += plus_or_dot(pieces) + rendered += "g%s" % pieces["short"] + else: + # exception #1 + rendered = "0.post%d" % pieces["distance"] + if pieces["dirty"]: + rendered += ".dev0" + rendered += "+g%s" % pieces["short"] + return rendered + + +def render_pep440_old(pieces): + """TAG[.postDISTANCE[.dev0]] . + + The ".dev0" means dirty. + + Eexceptions: + 1: no tags. 0.postDISTANCE[.dev0] + """ + if pieces["closest-tag"]: + rendered = pieces["closest-tag"] + if pieces["distance"] or pieces["dirty"]: + rendered += ".post%d" % pieces["distance"] + if pieces["dirty"]: + rendered += ".dev0" + else: + # exception #1 + rendered = "0.post%d" % pieces["distance"] + if pieces["dirty"]: + rendered += ".dev0" + return rendered + + +def render_git_describe(pieces): + """TAG[-DISTANCE-gHEX][-dirty]. + + Like 'git describe --tags --dirty --always'. + + Exceptions: + 1: no tags. HEX[-dirty] (note: no 'g' prefix) + """ + if pieces["closest-tag"]: + rendered = pieces["closest-tag"] + if pieces["distance"]: + rendered += "-%d-g%s" % (pieces["distance"], pieces["short"]) + else: + # exception #1 + rendered = pieces["short"] + if pieces["dirty"]: + rendered += "-dirty" + return rendered + + +def render_git_describe_long(pieces): + """TAG-DISTANCE-gHEX[-dirty]. + + Like 'git describe --tags --dirty --always -long'. + The distance/hash is unconditional. + + Exceptions: + 1: no tags. HEX[-dirty] (note: no 'g' prefix) + """ + if pieces["closest-tag"]: + rendered = pieces["closest-tag"] + rendered += "-%d-g%s" % (pieces["distance"], pieces["short"]) + else: + # exception #1 + rendered = pieces["short"] + if pieces["dirty"]: + rendered += "-dirty" + return rendered + + +def render(pieces, style): + """Render the given version pieces into the requested style.""" + if pieces["error"]: + return {"version": "unknown", + "full-revisionid": pieces.get("long"), + "dirty": None, + "error": pieces["error"]} + + if not style or style == "default": + style = "pep440" # the default + + if style == "pep440": + rendered = render_pep440(pieces) + elif style == "pep440-pre": + rendered = render_pep440_pre(pieces) + elif style == "pep440-post": + rendered = render_pep440_post(pieces) + elif style == "pep440-old": + rendered = render_pep440_old(pieces) + elif style == "git-describe": + rendered = render_git_describe(pieces) + elif style == "git-describe-long": + rendered = render_git_describe_long(pieces) + else: + raise ValueError("unknown style '%s'" % style) + + return {"version": rendered, "full-revisionid": pieces["long"], + "dirty": pieces["dirty"], "error": None} + + +def get_versions(): + """Get version information or return default if unable to do so.""" + # I am in _version.py, which lives at ROOT/VERSIONFILE_SOURCE. If we have + # __file__, we can work backwards from there to the root. Some + # py2exe/bbfreeze/non-CPython implementations don't do __file__, in which + # case we can only use expanded keywords. + + cfg = get_config() + verbose = cfg.verbose + + try: + return git_versions_from_keywords(get_keywords(), cfg.tag_prefix, + verbose) + except NotThisMethod: + pass + + try: + root = os.path.realpath(__file__) + # versionfile_source is the relative path from the top of the source + # tree (where the .git directory might live) to this file. Invert + # this to find the root from __file__. + for i in cfg.versionfile_source.split('/'): + root = os.path.dirname(root) + except NameError: + return {"version": "0+unknown", "full-revisionid": None, + "dirty": None, + "error": "unable to find root of source tree"} + + try: + pieces = git_pieces_from_vcs(cfg.tag_prefix, root, verbose) + return render(pieces, cfg.style) + except NotThisMethod: + pass + + try: + if cfg.parentdir_prefix: + return versions_from_parentdir(cfg.parentdir_prefix, root, verbose) + except NotThisMethod: + pass + + return {"version": "0+unknown", "full-revisionid": None, + "dirty": None, + "error": "unable to compute version"} diff --git a/geopandas/datasets/__init__.py b/geopandas/datasets/__init__.py new file mode 100644 index 0000000..35b24e0 --- /dev/null +++ b/geopandas/datasets/__init__.py @@ -0,0 +1,27 @@ +import os + + +__all__ = ['available', 'get_path'] + +module_path = os.path.dirname(__file__) +available = [p for p in next(os.walk(module_path))[1] + if not p.startswith('__')] + + +def get_path(dataset): + """ + Get the path to the data file. + + Parameters + ---------- + dataset : str + The name of the dataset. See ``geopandas.datasets.available`` for + all options. + + """ + if dataset in available: + return os.path.abspath( + os.path.join(module_path, dataset, dataset + '.shp')) + else: + msg = "The dataset '{data}' is not available".format(data=dataset) + raise ValueError(msg) diff --git a/doc/source/_example_data/naturalearth_cities.README.html b/geopandas/datasets/naturalearth_cities/naturalearth_cities.README.html similarity index 100% rename from doc/source/_example_data/naturalearth_cities.README.html rename to geopandas/datasets/naturalearth_cities/naturalearth_cities.README.html diff --git a/doc/source/_example_data/naturalearth_cities.VERSION.txt b/geopandas/datasets/naturalearth_cities/naturalearth_cities.VERSION.txt similarity index 100% rename from doc/source/_example_data/naturalearth_cities.VERSION.txt rename to geopandas/datasets/naturalearth_cities/naturalearth_cities.VERSION.txt diff --git a/doc/source/_example_data/naturalearth_cities.cpg b/geopandas/datasets/naturalearth_cities/naturalearth_cities.cpg similarity index 100% rename from doc/source/_example_data/naturalearth_cities.cpg rename to geopandas/datasets/naturalearth_cities/naturalearth_cities.cpg diff --git a/doc/source/_example_data/naturalearth_cities.dbf b/geopandas/datasets/naturalearth_cities/naturalearth_cities.dbf similarity index 100% rename from doc/source/_example_data/naturalearth_cities.dbf rename to geopandas/datasets/naturalearth_cities/naturalearth_cities.dbf diff --git a/doc/source/_example_data/naturalearth_cities.prj b/geopandas/datasets/naturalearth_cities/naturalearth_cities.prj similarity index 100% rename from doc/source/_example_data/naturalearth_cities.prj rename to geopandas/datasets/naturalearth_cities/naturalearth_cities.prj diff --git a/doc/source/_example_data/naturalearth_cities.shp b/geopandas/datasets/naturalearth_cities/naturalearth_cities.shp similarity index 100% rename from doc/source/_example_data/naturalearth_cities.shp rename to geopandas/datasets/naturalearth_cities/naturalearth_cities.shp diff --git a/doc/source/_example_data/naturalearth_cities.shx b/geopandas/datasets/naturalearth_cities/naturalearth_cities.shx similarity index 100% rename from doc/source/_example_data/naturalearth_cities.shx rename to geopandas/datasets/naturalearth_cities/naturalearth_cities.shx diff --git a/doc/source/_example_data/naturalearth_lowres.cpg b/geopandas/datasets/naturalearth_lowres/naturalearth_lowres.cpg similarity index 100% rename from doc/source/_example_data/naturalearth_lowres.cpg rename to geopandas/datasets/naturalearth_lowres/naturalearth_lowres.cpg diff --git a/doc/source/_example_data/naturalearth_lowres.dbf b/geopandas/datasets/naturalearth_lowres/naturalearth_lowres.dbf similarity index 100% rename from doc/source/_example_data/naturalearth_lowres.dbf rename to geopandas/datasets/naturalearth_lowres/naturalearth_lowres.dbf diff --git a/doc/source/_example_data/naturalearth_lowres.prj b/geopandas/datasets/naturalearth_lowres/naturalearth_lowres.prj similarity index 100% rename from doc/source/_example_data/naturalearth_lowres.prj rename to geopandas/datasets/naturalearth_lowres/naturalearth_lowres.prj diff --git a/doc/source/_example_data/naturalearth_lowres.shp b/geopandas/datasets/naturalearth_lowres/naturalearth_lowres.shp similarity index 100% rename from doc/source/_example_data/naturalearth_lowres.shp rename to geopandas/datasets/naturalearth_lowres/naturalearth_lowres.shp diff --git a/doc/source/_example_data/naturalearth_lowres.shx b/geopandas/datasets/naturalearth_lowres/naturalearth_lowres.shx similarity index 100% rename from doc/source/_example_data/naturalearth_lowres.shx rename to geopandas/datasets/naturalearth_lowres/naturalearth_lowres.shx diff --git a/geopandas/geodataframe.py b/geopandas/geodataframe.py index 78ec62b..3f0817d 100644 --- a/geopandas/geodataframe.py +++ b/geopandas/geodataframe.py @@ -8,7 +8,7 @@ import os import sys import numpy as np -from pandas import DataFrame, Series +from pandas import DataFrame, Series, Index from shapely.geometry import mapping, shape from shapely.geometry.base import BaseGeometry from six import string_types @@ -125,7 +125,7 @@ class GeoDataFrame(GeoPandasBase, DataFrame): crs = getattr(col, 'crs', self.crs) to_remove = None - geo_column_name = DEFAULT_GEO_COLUMN_NAME + geo_column_name = self._geometry_column_name if isinstance(col, (Series, list, np.ndarray)): level = col elif hasattr(col, 'ndim') and col.ndim != 1: @@ -139,7 +139,7 @@ class GeoDataFrame(GeoPandasBase, DataFrame): raise if drop: to_remove = col - geo_column_name = DEFAULT_GEO_COLUMN_NAME + geo_column_name = self._geometry_column_name else: geo_column_name = col @@ -409,10 +409,18 @@ class GeoDataFrame(GeoPandasBase, DataFrame): return GeoDataFrame def __finalize__(self, other, method=None, **kwargs): - """ propagate metadata from other to self """ - # NOTE: backported from pandas master (upcoming v0.13) - for name in self._metadata: - object.__setattr__(self, name, getattr(other, name, None)) + """propagate metadata from other to self """ + # merge operation: using metadata of the left object + if method == 'merge': + for name in self._metadata: + object.__setattr__(self, name, getattr(other.left, name, None)) + # concat operation: using metadata of the first object + elif method == 'concat': + for name in self._metadata: + object.__setattr__(self, name, getattr(other.objs[0], name, None)) + else: + for name in self._metadata: + object.__setattr__(self, name, getattr(other, name, None)) return self def copy(self, deep=True): @@ -441,6 +449,53 @@ class GeoDataFrame(GeoPandasBase, DataFrame): plot.__doc__ = plot_dataframe.__doc__ + def dissolve(self, by=None, aggfunc='first', as_index=True): + """ + Dissolve geometries within `groupby` into single observation. + This is accomplished by applying the `unary_union` method + to all geometries within a groupself. + + Observations associated with each `groupby` group will be aggregated + using the `aggfunc`. + + Parameters + ---------- + by : string, default None + Column whose values define groups to be dissolved + aggfunc : function or string, default "first" + Aggregation function for manipulation of data associated + with each group. Passed to pandas `groupby.agg` method. + as_index : boolean, default True + If true, groupby columns become index of result. + + Returns + ------- + GeoDataFrame + """ + + # Process non-spatial component + data = self.drop(labels=self.geometry.name, axis=1) + aggregated_data = data.groupby(by=by).agg(aggfunc) + + + # Process spatial component + def merge_geometries(block): + merged_geom = block.unary_union + return merged_geom + + g = self.groupby(by=by, group_keys=False)[self.geometry.name].agg(merge_geometries) + + # Aggregate + aggregated_geometry = GeoDataFrame(g, geometry=self.geometry.name) + # Recombine + aggregated = aggregated_geometry.join(aggregated_data) + + # Reset if requested + if not as_index: + aggregated = aggregated.reset_index() + + return aggregated + def _dataframe_set_geometry(self, col, drop=False, inplace=False, crs=None): if inplace: raise ValueError("Can't do inplace setting when converting from" diff --git a/geopandas/tests/test_dissolve.py b/geopandas/tests/test_dissolve.py new file mode 100644 index 0000000..45cc877 --- /dev/null +++ b/geopandas/tests/test_dissolve.py @@ -0,0 +1,83 @@ +from __future__ import absolute_import +import tempfile +import shutil +import numpy as np +from shapely.geometry import Point +from geopandas import GeoDataFrame, read_file +from geopandas.tools import overlay +from .util import unittest, download_nybb +from pandas.util.testing import assert_frame_equal +from pandas import Index +from distutils.version import LooseVersion +import pandas as pd + +pandas_0_15_problem = 'fails under pandas < 0.16 due to issue 324,'\ + 'not problem with dissolve.' + +class TestDataFrame(unittest.TestCase): + + def setUp(self): + + nybb_filename, nybb_zip_path = download_nybb() + self.polydf = read_file(nybb_zip_path, vfs='zip://' + nybb_filename) + self.polydf = self.polydf[['geometry', 'BoroName', 'BoroCode']] + + self.polydf = self.polydf.rename(columns={'geometry':'myshapes'}) + self.polydf = self.polydf.set_geometry('myshapes') + + self.polydf['manhattan_bronx'] = 5 + self.polydf.loc[3:4,'manhattan_bronx']=6 + + # Merged geometry + manhattan_bronx = self.polydf.loc[3:4,] + others = self.polydf.loc[0:2,] + + collapsed = [others.geometry.unary_union, manhattan_bronx.geometry.unary_union] + merged_shapes = GeoDataFrame({'myshapes': collapsed}, geometry='myshapes', + index=Index([5,6], name='manhattan_bronx')) + + # Different expected results + self.first = merged_shapes.copy() + self.first['BoroName'] = ['Staten Island', 'Manhattan'] + self.first['BoroCode'] = [5, 1] + + self.mean = merged_shapes.copy() + self.mean['BoroCode'] = [4,1.5] + + + @unittest.skipIf(str(pd.__version__) < LooseVersion('0.16'), pandas_0_15_problem) + def test_geom_dissolve(self): + test = self.polydf.dissolve('manhattan_bronx') + self.assertTrue(test.geometry.name == 'myshapes') + self.assertTrue(test.geom_almost_equals(self.first).all()) + + @unittest.skipIf(str(pd.__version__) < LooseVersion('0.16'), pandas_0_15_problem) + def test_first_dissolve(self): + test = self.polydf.dissolve('manhattan_bronx') + assert_frame_equal(self.first, test, check_column_type=False) + + @unittest.skipIf(str(pd.__version__) < LooseVersion('0.16'), pandas_0_15_problem) + def test_mean_dissolve(self): + test = self.polydf.dissolve('manhattan_bronx', aggfunc='mean') + assert_frame_equal(self.mean, test, check_column_type=False) + + test = self.polydf.dissolve('manhattan_bronx', aggfunc=np.mean) + assert_frame_equal(self.mean, test, check_column_type=False) + + @unittest.skipIf(str(pd.__version__) < LooseVersion('0.16'), pandas_0_15_problem) + def test_multicolumn_dissolve(self): + multi = self.polydf.copy() + multi['dup_col'] = multi.manhattan_bronx + multi_test = multi.dissolve(['manhattan_bronx', 'dup_col'], aggfunc='first') + + first = self.first.copy() + first['dup_col'] = first.index + first = first.set_index([first.index, 'dup_col']) + + assert_frame_equal(multi_test, first, check_column_type=False) + + @unittest.skipIf(str(pd.__version__) < LooseVersion('0.16'), pandas_0_15_problem) + def test_reset_index(self): + test = self.polydf.dissolve('manhattan_bronx', as_index=False) + comparison = self.first.reset_index() + assert_frame_equal(comparison, test, check_column_type=False) diff --git a/geopandas/tests/test_geodataframe.py b/geopandas/tests/test_geodataframe.py index 2796841..6739cd2 100644 --- a/geopandas/tests/test_geodataframe.py +++ b/geopandas/tests/test_geodataframe.py @@ -55,16 +55,12 @@ class TestDataFrame(unittest.TestCase): geom2 = [Point(x, y) for x, y in zip(range(5, 10), range(5))] df2 = df.set_geometry(geom2, crs='dummy_crs') - self.assert_('geometry' in df2) self.assert_('location' in df2) self.assertEqual(df2.crs, 'dummy_crs') self.assertEqual(df2.geometry.crs, 'dummy_crs') # reset so it outputs okay df2.crs = df.crs assert_geoseries_equal(df2.geometry, GeoSeries(geom2, crs=df2.crs)) - # for right now, non-geometry comes back as series - assert_geoseries_equal(df2['location'], df['location'], - check_series_type=False, check_dtype=False) def test_geo_getitem(self): data = {"A": range(5), "B": range(-5, 0), @@ -372,6 +368,18 @@ class TestDataFrame(unittest.TestCase): utm = lonlat.to_crs(epsg=26918) self.assertTrue(all(df2['geometry'].geom_almost_equals(utm['geometry'], decimal=2))) + def test_to_crs_geo_column_name(self): + # Test to_crs() with different geometry column name (GH#339) + df2 = self.df2.copy() + df2.crs = {'init': 'epsg:26918', 'no_defs': True} + df2 = df2.rename(columns={'geometry': 'geom'}) + df2.set_geometry('geom', inplace=True) + lonlat = df2.to_crs(epsg=4326) + utm = lonlat.to_crs(epsg=26918) + self.assertEqual(lonlat.geometry.name, 'geom') + self.assertEqual(utm.geometry.name, 'geom') + self.assertTrue(all(df2.geometry.geom_almost_equals(utm.geometry, decimal=2))) + def test_from_features(self): nybb_filename, nybb_zip_path = download_nybb() with fiona.open(nybb_zip_path, diff --git a/geopandas/tests/test_merge.py b/geopandas/tests/test_merge.py new file mode 100644 index 0000000..17cd238 --- /dev/null +++ b/geopandas/tests/test_merge.py @@ -0,0 +1,62 @@ +from __future__ import absolute_import + +import pandas as pd +from shapely.geometry import Point + +from geopandas import GeoDataFrame, GeoSeries +from geopandas.tests.util import unittest + + +class TestMerging(unittest.TestCase): + + def setUp(self): + + self.gseries = GeoSeries([Point(i, i) for i in range(3)]) + self.series = pd.Series([1, 2, 3]) + self.gdf = GeoDataFrame({'geometry': self.gseries, 'values': range(3)}) + self.df = pd.DataFrame({'col1': [1, 2, 3], 'col2': [0.1, 0.2, 0.3]}) + + def _check_metadata(self, gdf, geometry_column_name='geometry', crs=None): + + self.assertEqual(gdf._geometry_column_name, geometry_column_name) + self.assertEqual(gdf.crs, crs) + + def test_merge(self): + + res = self.gdf.merge(self.df, left_on='values', right_on='col1') + + # check result is a GeoDataFrame + self.assert_(isinstance(res, GeoDataFrame)) + + # check geometry property gives GeoSeries + self.assert_(isinstance(res.geometry, GeoSeries)) + + # check metadata + self._check_metadata(res) + + ## test that crs and other geometry name are preserved + self.gdf.crs = {'init' :'epsg:4326'} + self.gdf = (self.gdf.rename(columns={'geometry': 'points'}) + .set_geometry('points')) + res = self.gdf.merge(self.df, left_on='values', right_on='col1') + self.assert_(isinstance(res, GeoDataFrame)) + self.assert_(isinstance(res.geometry, GeoSeries)) + self._check_metadata(res, 'points', self.gdf.crs) + + def test_concat_axis0(self): + + res = pd.concat([self.gdf, self.gdf]) + + self.assertEqual(res.shape, (6, 2)) + self.assert_(isinstance(res, GeoDataFrame)) + self.assert_(isinstance(res.geometry, GeoSeries)) + self._check_metadata(res) + + def test_concat_axis1(self): + + res = pd.concat([self.gdf, self.df], axis=1) + + self.assertEqual(res.shape, (3, 4)) + self.assert_(isinstance(res, GeoDataFrame)) + self.assert_(isinstance(res.geometry, GeoSeries)) + self._check_metadata(res) diff --git a/geopandas/tests/test_overlay.py b/geopandas/tests/test_overlay.py index 0586fac..051618a 100644 --- a/geopandas/tests/test_overlay.py +++ b/geopandas/tests/test_overlay.py @@ -6,8 +6,8 @@ import shutil from shapely.geometry import Point from geopandas import GeoDataFrame, read_file -from geopandas.tools import overlay from geopandas.tests.util import unittest, download_nybb +from geopandas import overlay class TestDataFrame(unittest.TestCase): @@ -86,6 +86,28 @@ class TestDataFrame(unittest.TestCase): df = overlay(self.polydf, polydf2r, how="union") self.assertTrue('Shape_Area_2' in df.columns and 'Shape_Area' in df.columns) + def test_geometry_not_named_geometry(self): + # Issue #306 + # Add points and flip names + polydf3 = self.polydf.copy() + polydf3 = polydf3.rename(columns={'geometry':'polygons'}) + polydf3 = polydf3.set_geometry('polygons') + polydf3['geometry'] = self.pointdf.geometry.loc[0:4] + self.assertTrue(polydf3.geometry.name == 'polygons') + + df = overlay(polydf3, self.polydf2, how="union") + self.assertTrue(type(df) is GeoDataFrame) + + df2 = overlay(self.polydf, self.polydf2, how="union") + self.assertTrue(df.geom_almost_equals(df2).all()) + + def test_geoseries_warning(self): + # Issue #305 + + def f(): + overlay(self.polydf, self.polydf2.geometry, how="union") + self.assertRaises(NotImplementedError, f) + diff --git a/geopandas/tools/overlay.py b/geopandas/tools/overlay.py index 4d14cd1..8abbca7 100644 --- a/geopandas/tools/overlay.py +++ b/geopandas/tools/overlay.py @@ -17,7 +17,7 @@ def _uniquify(columns): def _extract_rings(df): - """Collects all inner and outer linear rings from a GeoDataFrame + """Collects all inner and outer linear rings from a GeoDataFrame with (multi)Polygon geometeries Parameters @@ -30,8 +30,10 @@ def _extract_rings(df): """ poly_msg = "overlay only takes GeoDataFrames with (multi)polygon geometries" rings = [] + geometry_column = df.geometry.name + for i, feat in df.iterrows(): - geom = feat.geometry + geom = feat[geometry_column] if geom.type not in ['Polygon', 'MultiPolygon']: raise TypeError(poly_msg) @@ -53,22 +55,28 @@ def _extract_rings(df): return rings def overlay(df1, df2, how, use_sindex=True): - """Perform spatial overlay between two polygons - Currently only supports data GeoDataFrames with polygons + """Perform spatial overlay between two polygons. - Implements several methods (see `allowed_hows` list) that are - all effectively subsets of the union. + Currently only supports data GeoDataFrames with polygons. + Implements several methods that are all effectively subsets of + the union. Parameters ---------- df1 : GeoDataFrame with MultiPolygon or Polygon geometry column df2 : GeoDataFrame with MultiPolygon or Polygon geometry column - how : method of spatial overlay - use_sindex : Boolean; Use the spatial index to speed up operation. Default is True. + how : string + Method of spatial overlay: 'intersection', 'union', + 'identity', 'symmetric_difference' or 'difference'. + use_sindex : boolean, default True + Use the spatial index to speed up operation if available. Returns ------- - df : GeoDataFrame with new set of polygons and attributes resulting from the overlay + df : GeoDataFrame + GeoDataFrame with new set of polygons and attributes + resulting from the overlay + """ allowed_hows = [ 'intersection', @@ -82,6 +90,9 @@ def overlay(df1, df2, how, use_sindex=True): raise ValueError("`how` was \"%s\" but is expected to be in %s" % \ (how, allowed_hows)) + if isinstance(df1, GeoSeries) or isinstance(df2, GeoSeries): + raise NotImplementedError("overlay currently only implemented for GeoDataFrames") + # Collect the interior and exterior rings rings1 = _extract_rings(df1) rings2 = _extract_rings(df2) @@ -125,13 +136,13 @@ def overlay(df1, df2, how, use_sindex=True): prop2 = None for cand_id in candidates1: cand = df1.ix[cand_id] - if cent.intersects(cand.geometry): + if cent.intersects(cand[df1.geometry.name]): df1_hit = True prop1 = cand break # Take the first hit for cand_id in candidates2: cand = df2.ix[cand_id] - if cent.intersects(cand.geometry): + if cent.intersects(cand[df2.geometry.name]): df2_hit = True prop2 = cand break # Take the first hit diff --git a/geopandas/tools/sjoin.py b/geopandas/tools/sjoin.py index ba6c7cc..206f4cb 100644 --- a/geopandas/tools/sjoin.py +++ b/geopandas/tools/sjoin.py @@ -4,21 +4,28 @@ from shapely import prepared def sjoin(left_df, right_df, how='inner', op='intersects', - lsuffix='left', rsuffix='right', **kwargs): + lsuffix='left', rsuffix='right'): """Spatial join of two GeoDataFrames. - left_df, right_df are GeoDataFrames - how: type of join - left -> use keys from left_df; retain only left_df geometry column - right -> use keys from right_df; retain only right_df geometry column - inner -> use intersection of keys from both dfs; - retain only left_df geometry column - op: binary predicate {'intersects', 'contains', 'within'} - see http://toblerity.org/shapely/manual.html#binary-predicates - lsuffix: suffix to apply to overlapping column names (left GeoDataFrame) - rsuffix: suffix to apply to overlapping column names (right GeoDataFrame) - """ + Parameters + ---------- + left_df, right_df : GeoDataFrames + how : string, default 'inner' + The type of join: + * 'left': use keys from left_df; retain only left_df geometry column + * 'right': use keys from right_df; retain only right_df geometry column + * 'inner': use intersection of keys from both dfs; retain only + left_df geometry column + op : string, default 'intersection' + Binary predicate, one of {'intersects', 'contains', 'within'}. + See http://toblerity.org/shapely/manual.html#binary-predicates. + lsuffix : string, default 'left' + Suffix to apply to overlapping column names (left GeoDataFrame). + rsuffix : string, default 'right' + Suffix to apply to overlapping column names (right GeoDataFrame). + + """ import rtree allowed_hows = ['left', 'right', 'inner'] diff --git a/geopandas/tools/tests/test_sjoin.py b/geopandas/tools/tests/test_sjoin.py index c3b85dd..cfaaf37 100644 --- a/geopandas/tools/tests/test_sjoin.py +++ b/geopandas/tools/tests/test_sjoin.py @@ -8,7 +8,7 @@ from shapely.geometry import Point from geopandas import GeoDataFrame, read_file, base from geopandas.tests.util import unittest, download_nybb -from geopandas.tools import sjoin +from geopandas import sjoin @unittest.skipIf(not base.HAS_SINDEX, 'Rtree absent, skipping') diff --git a/readthedocs.yml b/readthedocs.yml index 1c401c1..50b41dd 100644 --- a/readthedocs.yml +++ b/readthedocs.yml @@ -1,3 +1,5 @@ +formats: + - none conda: file: doc/environment.yml python: diff --git a/setup.cfg b/setup.cfg index 2a9acf1..96b08c0 100644 --- a/setup.cfg +++ b/setup.cfg @@ -1,2 +1,10 @@ [bdist_wheel] universal = 1 + +[versioneer] +VCS = git +style = pep440 +versionfile_source = geopandas/_version.py +versionfile_build = geopandas/_version.py +tag_prefix = v +parentdir_prefix = geopandas- diff --git a/setup.py b/setup.py index 7dc914b..3d2bed0 100644 --- a/setup.py +++ b/setup.py @@ -1,18 +1,17 @@ #!/usr/bin/env/python """Installation script -Version handling borrowed from pandas project. """ -import sys import os -import warnings try: from setuptools import setup except ImportError: from distutils.core import setup +import versioneer + LONG_DESCRIPTION = """GeoPandas is a project to add support for geographic data to `pandas`_ objects. @@ -27,67 +26,28 @@ such as PostGIS. .. _shapely: http://toblerity.github.io/shapely """ -MAJOR = 0 -MINOR = 1 -MICRO = 0 -ISRELEASED = False -VERSION = '%d.%d.%d' % (MAJOR, MINOR, MICRO) -QUALIFIER = '' if os.environ.get('READTHEDOCS', False) == 'True': INSTALL_REQUIRES = [] else: INSTALL_REQUIRES = ['pandas', 'shapely', 'fiona', 'descartes', 'pyproj'] -FULLVERSION = VERSION -if not ISRELEASED: - FULLVERSION += '.dev' - try: - import subprocess - try: - pipe = subprocess.Popen(["git", "rev-parse", "--short", "HEAD"], - stdout=subprocess.PIPE).stdout - except OSError: - # msysgit compatibility - pipe = subprocess.Popen( - ["git.cmd", "describe", "HEAD"], - stdout=subprocess.PIPE).stdout - rev = pipe.read().strip() - # makes distutils blow up on Python 2.7 - if sys.version_info[0] >= 3: - rev = rev.decode('ascii') +# get all data dirs in the datasets module +data_files = [] - FULLVERSION = '%d.%d.%d.dev-%s' % (MAJOR, MINOR, MICRO, rev) - - except: - warnings.warn("WARNING: Couldn't get git revision") -else: - FULLVERSION += QUALIFIER - - -def write_version_py(filename=None): - cnt = """\ -version = '%s' -short_version = '%s' -""" - if not filename: - filename = os.path.join( - os.path.dirname(__file__), 'geopandas', 'version.py') - - a = open(filename, 'w') - try: - a.write(cnt % (FULLVERSION, VERSION)) - finally: - a.close() - -write_version_py() +for item in os.listdir("geopandas/datasets"): + if os.path.isdir(os.path.join("geopandas/datasets/", item)) \ + and not item.startswith('__'): + data_files.append(os.path.join("datasets", item, '*')) setup(name='geopandas', - version=FULLVERSION, + version=versioneer.get_version(), description='Geographic pandas extensions', license='BSD', - author='Kelsey Jordahl', - author_email='kjordahl@enthought.com', + author='GeoPandas contributors', + author_email='kjordahl@alum.mit.edu', url='http://geopandas.org', long_description=LONG_DESCRIPTION, - packages=['geopandas', 'geopandas.io', 'geopandas.tools'], + packages=['geopandas', 'geopandas.io', 'geopandas.tools', + 'geopandas.datasets'], + package_data={'geopandas': data_files}, install_requires=INSTALL_REQUIRES) diff --git a/versioneer.py b/versioneer.py new file mode 100644 index 0000000..7ed2a21 --- /dev/null +++ b/versioneer.py @@ -0,0 +1,1774 @@ + +# Version: 0.16 + +"""The Versioneer - like a rocketeer, but for versions. + +The Versioneer +============== + +* like a rocketeer, but for versions! +* https://github.com/warner/python-versioneer +* Brian Warner +* License: Public Domain +* Compatible With: python2.6, 2.7, 3.3, 3.4, 3.5, and pypy +* [![Latest Version] +(https://pypip.in/version/versioneer/badge.svg?style=flat) +](https://pypi.python.org/pypi/versioneer/) +* [![Build Status] +(https://travis-ci.org/warner/python-versioneer.png?branch=master) +](https://travis-ci.org/warner/python-versioneer) + +This is a tool for managing a recorded version number in distutils-based +python projects. The goal is to remove the tedious and error-prone "update +the embedded version string" step from your release process. Making a new +release should be as easy as recording a new tag in your version-control +system, and maybe making new tarballs. + + +## Quick Install + +* `pip install versioneer` to somewhere to your $PATH +* add a `[versioneer]` section to your setup.cfg (see below) +* run `versioneer install` in your source tree, commit the results + +## Version Identifiers + +Source trees come from a variety of places: + +* a version-control system checkout (mostly used by developers) +* a nightly tarball, produced by build automation +* a snapshot tarball, produced by a web-based VCS browser, like github's + "tarball from tag" feature +* a release tarball, produced by "setup.py sdist", distributed through PyPI + +Within each source tree, the version identifier (either a string or a number, +this tool is format-agnostic) can come from a variety of places: + +* ask the VCS tool itself, e.g. "git describe" (for checkouts), which knows + about recent "tags" and an absolute revision-id +* the name of the directory into which the tarball was unpacked +* an expanded VCS keyword ($Id$, etc) +* a `_version.py` created by some earlier build step + +For released software, the version identifier is closely related to a VCS +tag. Some projects use tag names that include more than just the version +string (e.g. "myproject-1.2" instead of just "1.2"), in which case the tool +needs to strip the tag prefix to extract the version identifier. For +unreleased software (between tags), the version identifier should provide +enough information to help developers recreate the same tree, while also +giving them an idea of roughly how old the tree is (after version 1.2, before +version 1.3). Many VCS systems can report a description that captures this, +for example `git describe --tags --dirty --always` reports things like +"0.7-1-g574ab98-dirty" to indicate that the checkout is one revision past the +0.7 tag, has a unique revision id of "574ab98", and is "dirty" (it has +uncommitted changes. + +The version identifier is used for multiple purposes: + +* to allow the module to self-identify its version: `myproject.__version__` +* to choose a name and prefix for a 'setup.py sdist' tarball + +## Theory of Operation + +Versioneer works by adding a special `_version.py` file into your source +tree, where your `__init__.py` can import it. This `_version.py` knows how to +dynamically ask the VCS tool for version information at import time. + +`_version.py` also contains `$Revision$` markers, and the installation +process marks `_version.py` to have this marker rewritten with a tag name +during the `git archive` command. As a result, generated tarballs will +contain enough information to get the proper version. + +To allow `setup.py` to compute a version too, a `versioneer.py` is added to +the top level of your source tree, next to `setup.py` and the `setup.cfg` +that configures it. This overrides several distutils/setuptools commands to +compute the version when invoked, and changes `setup.py build` and `setup.py +sdist` to replace `_version.py` with a small static file that contains just +the generated version data. + +## Installation + +First, decide on values for the following configuration variables: + +* `VCS`: the version control system you use. Currently accepts "git". + +* `style`: the style of version string to be produced. See "Styles" below for + details. Defaults to "pep440", which looks like + `TAG[+DISTANCE.gSHORTHASH[.dirty]]`. + +* `versionfile_source`: + + A project-relative pathname into which the generated version strings should + be written. This is usually a `_version.py` next to your project's main + `__init__.py` file, so it can be imported at runtime. If your project uses + `src/myproject/__init__.py`, this should be `src/myproject/_version.py`. + This file should be checked in to your VCS as usual: the copy created below + by `setup.py setup_versioneer` will include code that parses expanded VCS + keywords in generated tarballs. The 'build' and 'sdist' commands will + replace it with a copy that has just the calculated version string. + + This must be set even if your project does not have any modules (and will + therefore never import `_version.py`), since "setup.py sdist" -based trees + still need somewhere to record the pre-calculated version strings. Anywhere + in the source tree should do. If there is a `__init__.py` next to your + `_version.py`, the `setup.py setup_versioneer` command (described below) + will append some `__version__`-setting assignments, if they aren't already + present. + +* `versionfile_build`: + + Like `versionfile_source`, but relative to the build directory instead of + the source directory. These will differ when your setup.py uses + 'package_dir='. If you have `package_dir={'myproject': 'src/myproject'}`, + then you will probably have `versionfile_build='myproject/_version.py'` and + `versionfile_source='src/myproject/_version.py'`. + + If this is set to None, then `setup.py build` will not attempt to rewrite + any `_version.py` in the built tree. If your project does not have any + libraries (e.g. if it only builds a script), then you should use + `versionfile_build = None`. To actually use the computed version string, + your `setup.py` will need to override `distutils.command.build_scripts` + with a subclass that explicitly inserts a copy of + `versioneer.get_version()` into your script file. See + `test/demoapp-script-only/setup.py` for an example. + +* `tag_prefix`: + + a string, like 'PROJECTNAME-', which appears at the start of all VCS tags. + If your tags look like 'myproject-1.2.0', then you should use + tag_prefix='myproject-'. If you use unprefixed tags like '1.2.0', this + should be an empty string, using either `tag_prefix=` or `tag_prefix=''`. + +* `parentdir_prefix`: + + a optional string, frequently the same as tag_prefix, which appears at the + start of all unpacked tarball filenames. If your tarball unpacks into + 'myproject-1.2.0', this should be 'myproject-'. To disable this feature, + just omit the field from your `setup.cfg`. + +This tool provides one script, named `versioneer`. That script has one mode, +"install", which writes a copy of `versioneer.py` into the current directory +and runs `versioneer.py setup` to finish the installation. + +To versioneer-enable your project: + +* 1: Modify your `setup.cfg`, adding a section named `[versioneer]` and + populating it with the configuration values you decided earlier (note that + the option names are not case-sensitive): + + ```` + [versioneer] + VCS = git + style = pep440 + versionfile_source = src/myproject/_version.py + versionfile_build = myproject/_version.py + tag_prefix = + parentdir_prefix = myproject- + ```` + +* 2: Run `versioneer install`. This will do the following: + + * copy `versioneer.py` into the top of your source tree + * create `_version.py` in the right place (`versionfile_source`) + * modify your `__init__.py` (if one exists next to `_version.py`) to define + `__version__` (by calling a function from `_version.py`) + * modify your `MANIFEST.in` to include both `versioneer.py` and the + generated `_version.py` in sdist tarballs + + `versioneer install` will complain about any problems it finds with your + `setup.py` or `setup.cfg`. Run it multiple times until you have fixed all + the problems. + +* 3: add a `import versioneer` to your setup.py, and add the following + arguments to the setup() call: + + version=versioneer.get_version(), + cmdclass=versioneer.get_cmdclass(), + +* 4: commit these changes to your VCS. To make sure you won't forget, + `versioneer install` will mark everything it touched for addition using + `git add`. Don't forget to add `setup.py` and `setup.cfg` too. + +## Post-Installation Usage + +Once established, all uses of your tree from a VCS checkout should get the +current version string. All generated tarballs should include an embedded +version string (so users who unpack them will not need a VCS tool installed). + +If you distribute your project through PyPI, then the release process should +boil down to two steps: + +* 1: git tag 1.0 +* 2: python setup.py register sdist upload + +If you distribute it through github (i.e. users use github to generate +tarballs with `git archive`), the process is: + +* 1: git tag 1.0 +* 2: git push; git push --tags + +Versioneer will report "0+untagged.NUMCOMMITS.gHASH" until your tree has at +least one tag in its history. + +## Version-String Flavors + +Code which uses Versioneer can learn about its version string at runtime by +importing `_version` from your main `__init__.py` file and running the +`get_versions()` function. From the "outside" (e.g. in `setup.py`), you can +import the top-level `versioneer.py` and run `get_versions()`. + +Both functions return a dictionary with different flavors of version +information: + +* `['version']`: A condensed version string, rendered using the selected + style. This is the most commonly used value for the project's version + string. The default "pep440" style yields strings like `0.11`, + `0.11+2.g1076c97`, or `0.11+2.g1076c97.dirty`. See the "Styles" section + below for alternative styles. + +* `['full-revisionid']`: detailed revision identifier. For Git, this is the + full SHA1 commit id, e.g. "1076c978a8d3cfc70f408fe5974aa6c092c949ac". + +* `['dirty']`: a boolean, True if the tree has uncommitted changes. Note that + this is only accurate if run in a VCS checkout, otherwise it is likely to + be False or None + +* `['error']`: if the version string could not be computed, this will be set + to a string describing the problem, otherwise it will be None. It may be + useful to throw an exception in setup.py if this is set, to avoid e.g. + creating tarballs with a version string of "unknown". + +Some variants are more useful than others. Including `full-revisionid` in a +bug report should allow developers to reconstruct the exact code being tested +(or indicate the presence of local changes that should be shared with the +developers). `version` is suitable for display in an "about" box or a CLI +`--version` output: it can be easily compared against release notes and lists +of bugs fixed in various releases. + +The installer adds the following text to your `__init__.py` to place a basic +version in `YOURPROJECT.__version__`: + + from ._version import get_versions + __version__ = get_versions()['version'] + del get_versions + +## Styles + +The setup.cfg `style=` configuration controls how the VCS information is +rendered into a version string. + +The default style, "pep440", produces a PEP440-compliant string, equal to the +un-prefixed tag name for actual releases, and containing an additional "local +version" section with more detail for in-between builds. For Git, this is +TAG[+DISTANCE.gHEX[.dirty]] , using information from `git describe --tags +--dirty --always`. For example "0.11+2.g1076c97.dirty" indicates that the +tree is like the "1076c97" commit but has uncommitted changes (".dirty"), and +that this commit is two revisions ("+2") beyond the "0.11" tag. For released +software (exactly equal to a known tag), the identifier will only contain the +stripped tag, e.g. "0.11". + +Other styles are available. See details.md in the Versioneer source tree for +descriptions. + +## Debugging + +Versioneer tries to avoid fatal errors: if something goes wrong, it will tend +to return a version of "0+unknown". To investigate the problem, run `setup.py +version`, which will run the version-lookup code in a verbose mode, and will +display the full contents of `get_versions()` (including the `error` string, +which may help identify what went wrong). + +## Updating Versioneer + +To upgrade your project to a new release of Versioneer, do the following: + +* install the new Versioneer (`pip install -U versioneer` or equivalent) +* edit `setup.cfg`, if necessary, to include any new configuration settings + indicated by the release notes +* re-run `versioneer install` in your source tree, to replace + `SRC/_version.py` +* commit any changed files + +### Upgrading to 0.16 + +Nothing special. + +### Upgrading to 0.15 + +Starting with this version, Versioneer is configured with a `[versioneer]` +section in your `setup.cfg` file. Earlier versions required the `setup.py` to +set attributes on the `versioneer` module immediately after import. The new +version will refuse to run (raising an exception during import) until you +have provided the necessary `setup.cfg` section. + +In addition, the Versioneer package provides an executable named +`versioneer`, and the installation process is driven by running `versioneer +install`. In 0.14 and earlier, the executable was named +`versioneer-installer` and was run without an argument. + +### Upgrading to 0.14 + +0.14 changes the format of the version string. 0.13 and earlier used +hyphen-separated strings like "0.11-2-g1076c97-dirty". 0.14 and beyond use a +plus-separated "local version" section strings, with dot-separated +components, like "0.11+2.g1076c97". PEP440-strict tools did not like the old +format, but should be ok with the new one. + +### Upgrading from 0.11 to 0.12 + +Nothing special. + +### Upgrading from 0.10 to 0.11 + +You must add a `versioneer.VCS = "git"` to your `setup.py` before re-running +`setup.py setup_versioneer`. This will enable the use of additional +version-control systems (SVN, etc) in the future. + +## Future Directions + +This tool is designed to make it easily extended to other version-control +systems: all VCS-specific components are in separate directories like +src/git/ . The top-level `versioneer.py` script is assembled from these +components by running make-versioneer.py . In the future, make-versioneer.py +will take a VCS name as an argument, and will construct a version of +`versioneer.py` that is specific to the given VCS. It might also take the +configuration arguments that are currently provided manually during +installation by editing setup.py . Alternatively, it might go the other +direction and include code from all supported VCS systems, reducing the +number of intermediate scripts. + + +## License + +To make Versioneer easier to embed, all its code is dedicated to the public +domain. The `_version.py` that it creates is also in the public domain. +Specifically, both are released under the Creative Commons "Public Domain +Dedication" license (CC0-1.0), as described in +https://creativecommons.org/publicdomain/zero/1.0/ . + +""" + +from __future__ import print_function +try: + import configparser +except ImportError: + import ConfigParser as configparser +import errno +import json +import os +import re +import subprocess +import sys + + +class VersioneerConfig: + """Container for Versioneer configuration parameters.""" + + +def get_root(): + """Get the project root directory. + + We require that all commands are run from the project root, i.e. the + directory that contains setup.py, setup.cfg, and versioneer.py . + """ + root = os.path.realpath(os.path.abspath(os.getcwd())) + setup_py = os.path.join(root, "setup.py") + versioneer_py = os.path.join(root, "versioneer.py") + if not (os.path.exists(setup_py) or os.path.exists(versioneer_py)): + # allow 'python path/to/setup.py COMMAND' + root = os.path.dirname(os.path.realpath(os.path.abspath(sys.argv[0]))) + setup_py = os.path.join(root, "setup.py") + versioneer_py = os.path.join(root, "versioneer.py") + if not (os.path.exists(setup_py) or os.path.exists(versioneer_py)): + err = ("Versioneer was unable to run the project root directory. " + "Versioneer requires setup.py to be executed from " + "its immediate directory (like 'python setup.py COMMAND'), " + "or in a way that lets it use sys.argv[0] to find the root " + "(like 'python path/to/setup.py COMMAND').") + raise VersioneerBadRootError(err) + try: + # Certain runtime workflows (setup.py install/develop in a setuptools + # tree) execute all dependencies in a single python process, so + # "versioneer" may be imported multiple times, and python's shared + # module-import table will cache the first one. So we can't use + # os.path.dirname(__file__), as that will find whichever + # versioneer.py was first imported, even in later projects. + me = os.path.realpath(os.path.abspath(__file__)) + if os.path.splitext(me)[0] != os.path.splitext(versioneer_py)[0]: + print("Warning: build in %s is using versioneer.py from %s" + % (os.path.dirname(me), versioneer_py)) + except NameError: + pass + return root + + +def get_config_from_root(root): + """Read the project setup.cfg file to determine Versioneer config.""" + # This might raise EnvironmentError (if setup.cfg is missing), or + # configparser.NoSectionError (if it lacks a [versioneer] section), or + # configparser.NoOptionError (if it lacks "VCS="). See the docstring at + # the top of versioneer.py for instructions on writing your setup.cfg . + setup_cfg = os.path.join(root, "setup.cfg") + parser = configparser.SafeConfigParser() + with open(setup_cfg, "r") as f: + parser.readfp(f) + VCS = parser.get("versioneer", "VCS") # mandatory + + def get(parser, name): + if parser.has_option("versioneer", name): + return parser.get("versioneer", name) + return None + cfg = VersioneerConfig() + cfg.VCS = VCS + cfg.style = get(parser, "style") or "" + cfg.versionfile_source = get(parser, "versionfile_source") + cfg.versionfile_build = get(parser, "versionfile_build") + cfg.tag_prefix = get(parser, "tag_prefix") + if cfg.tag_prefix in ("''", '""'): + cfg.tag_prefix = "" + cfg.parentdir_prefix = get(parser, "parentdir_prefix") + cfg.verbose = get(parser, "verbose") + return cfg + + +class NotThisMethod(Exception): + """Exception raised if a method is not valid for the current scenario.""" + +# these dictionaries contain VCS-specific tools +LONG_VERSION_PY = {} +HANDLERS = {} + + +def register_vcs_handler(vcs, method): # decorator + """Decorator to mark a method as the handler for a particular VCS.""" + def decorate(f): + """Store f in HANDLERS[vcs][method].""" + if vcs not in HANDLERS: + HANDLERS[vcs] = {} + HANDLERS[vcs][method] = f + return f + return decorate + + +def run_command(commands, args, cwd=None, verbose=False, hide_stderr=False): + """Call the given command(s).""" + assert isinstance(commands, list) + p = None + for c in commands: + try: + dispcmd = str([c] + args) + # remember shell=False, so use git.cmd on windows, not just git + p = subprocess.Popen([c] + args, cwd=cwd, stdout=subprocess.PIPE, + stderr=(subprocess.PIPE if hide_stderr + else None)) + break + except EnvironmentError: + e = sys.exc_info()[1] + if e.errno == errno.ENOENT: + continue + if verbose: + print("unable to run %s" % dispcmd) + print(e) + return None + else: + if verbose: + print("unable to find command, tried %s" % (commands,)) + return None + stdout = p.communicate()[0].strip() + if sys.version_info[0] >= 3: + stdout = stdout.decode() + if p.returncode != 0: + if verbose: + print("unable to run %s (error)" % dispcmd) + return None + return stdout +LONG_VERSION_PY['git'] = ''' +# This file helps to compute a version number in source trees obtained from +# git-archive tarball (such as those provided by githubs download-from-tag +# feature). Distribution tarballs (built by setup.py sdist) and build +# directories (produced by setup.py build) will contain a much shorter file +# that just contains the computed version number. + +# This file is released into the public domain. Generated by +# versioneer-0.16 (https://github.com/warner/python-versioneer) + +"""Git implementation of _version.py.""" + +import errno +import os +import re +import subprocess +import sys + + +def get_keywords(): + """Get the keywords needed to look up the version information.""" + # these strings will be replaced by git during git-archive. + # setup.py/versioneer.py will grep for the variable names, so they must + # each be defined on a line of their own. _version.py will just call + # get_keywords(). + git_refnames = "%(DOLLAR)sFormat:%%d%(DOLLAR)s" + git_full = "%(DOLLAR)sFormat:%%H%(DOLLAR)s" + keywords = {"refnames": git_refnames, "full": git_full} + return keywords + + +class VersioneerConfig: + """Container for Versioneer configuration parameters.""" + + +def get_config(): + """Create, populate and return the VersioneerConfig() object.""" + # these strings are filled in when 'setup.py versioneer' creates + # _version.py + cfg = VersioneerConfig() + cfg.VCS = "git" + cfg.style = "%(STYLE)s" + cfg.tag_prefix = "%(TAG_PREFIX)s" + cfg.parentdir_prefix = "%(PARENTDIR_PREFIX)s" + cfg.versionfile_source = "%(VERSIONFILE_SOURCE)s" + cfg.verbose = False + return cfg + + +class NotThisMethod(Exception): + """Exception raised if a method is not valid for the current scenario.""" + + +LONG_VERSION_PY = {} +HANDLERS = {} + + +def register_vcs_handler(vcs, method): # decorator + """Decorator to mark a method as the handler for a particular VCS.""" + def decorate(f): + """Store f in HANDLERS[vcs][method].""" + if vcs not in HANDLERS: + HANDLERS[vcs] = {} + HANDLERS[vcs][method] = f + return f + return decorate + + +def run_command(commands, args, cwd=None, verbose=False, hide_stderr=False): + """Call the given command(s).""" + assert isinstance(commands, list) + p = None + for c in commands: + try: + dispcmd = str([c] + args) + # remember shell=False, so use git.cmd on windows, not just git + p = subprocess.Popen([c] + args, cwd=cwd, stdout=subprocess.PIPE, + stderr=(subprocess.PIPE if hide_stderr + else None)) + break + except EnvironmentError: + e = sys.exc_info()[1] + if e.errno == errno.ENOENT: + continue + if verbose: + print("unable to run %%s" %% dispcmd) + print(e) + return None + else: + if verbose: + print("unable to find command, tried %%s" %% (commands,)) + return None + stdout = p.communicate()[0].strip() + if sys.version_info[0] >= 3: + stdout = stdout.decode() + if p.returncode != 0: + if verbose: + print("unable to run %%s (error)" %% dispcmd) + return None + return stdout + + +def versions_from_parentdir(parentdir_prefix, root, verbose): + """Try to determine the version from the parent directory name. + + Source tarballs conventionally unpack into a directory that includes + both the project name and a version string. + """ + dirname = os.path.basename(root) + if not dirname.startswith(parentdir_prefix): + if verbose: + print("guessing rootdir is '%%s', but '%%s' doesn't start with " + "prefix '%%s'" %% (root, dirname, parentdir_prefix)) + raise NotThisMethod("rootdir doesn't start with parentdir_prefix") + return {"version": dirname[len(parentdir_prefix):], + "full-revisionid": None, + "dirty": False, "error": None} + + +@register_vcs_handler("git", "get_keywords") +def git_get_keywords(versionfile_abs): + """Extract version information from the given file.""" + # the code embedded in _version.py can just fetch the value of these + # keywords. When used from setup.py, we don't want to import _version.py, + # so we do it with a regexp instead. This function is not used from + # _version.py. + keywords = {} + try: + f = open(versionfile_abs, "r") + for line in f.readlines(): + if line.strip().startswith("git_refnames ="): + mo = re.search(r'=\s*"(.*)"', line) + if mo: + keywords["refnames"] = mo.group(1) + if line.strip().startswith("git_full ="): + mo = re.search(r'=\s*"(.*)"', line) + if mo: + keywords["full"] = mo.group(1) + f.close() + except EnvironmentError: + pass + return keywords + + +@register_vcs_handler("git", "keywords") +def git_versions_from_keywords(keywords, tag_prefix, verbose): + """Get version information from git keywords.""" + if not keywords: + raise NotThisMethod("no keywords at all, weird") + refnames = keywords["refnames"].strip() + if refnames.startswith("$Format"): + if verbose: + print("keywords are unexpanded, not using") + raise NotThisMethod("unexpanded keywords, not a git-archive tarball") + refs = set([r.strip() for r in refnames.strip("()").split(",")]) + # starting in git-1.8.3, tags are listed as "tag: foo-1.0" instead of + # just "foo-1.0". If we see a "tag: " prefix, prefer those. + TAG = "tag: " + tags = set([r[len(TAG):] for r in refs if r.startswith(TAG)]) + if not tags: + # Either we're using git < 1.8.3, or there really are no tags. We use + # a heuristic: assume all version tags have a digit. The old git %%d + # expansion behaves like git log --decorate=short and strips out the + # refs/heads/ and refs/tags/ prefixes that would let us distinguish + # between branches and tags. By ignoring refnames without digits, we + # filter out many common branch names like "release" and + # "stabilization", as well as "HEAD" and "master". + tags = set([r for r in refs if re.search(r'\d', r)]) + if verbose: + print("discarding '%%s', no digits" %% ",".join(refs-tags)) + if verbose: + print("likely tags: %%s" %% ",".join(sorted(tags))) + for ref in sorted(tags): + # sorting will prefer e.g. "2.0" over "2.0rc1" + if ref.startswith(tag_prefix): + r = ref[len(tag_prefix):] + if verbose: + print("picking %%s" %% r) + return {"version": r, + "full-revisionid": keywords["full"].strip(), + "dirty": False, "error": None + } + # no suitable tags, so version is "0+unknown", but full hex is still there + if verbose: + print("no suitable tags, using unknown + full revision id") + return {"version": "0+unknown", + "full-revisionid": keywords["full"].strip(), + "dirty": False, "error": "no suitable tags"} + + +@register_vcs_handler("git", "pieces_from_vcs") +def git_pieces_from_vcs(tag_prefix, root, verbose, run_command=run_command): + """Get version from 'git describe' in the root of the source tree. + + This only gets called if the git-archive 'subst' keywords were *not* + expanded, and _version.py hasn't already been rewritten with a short + version string, meaning we're inside a checked out source tree. + """ + if not os.path.exists(os.path.join(root, ".git")): + if verbose: + print("no .git in %%s" %% root) + raise NotThisMethod("no .git directory") + + GITS = ["git"] + if sys.platform == "win32": + GITS = ["git.cmd", "git.exe"] + # if there is a tag matching tag_prefix, this yields TAG-NUM-gHEX[-dirty] + # if there isn't one, this yields HEX[-dirty] (no NUM) + describe_out = run_command(GITS, ["describe", "--tags", "--dirty", + "--always", "--long", + "--match", "%%s*" %% tag_prefix], + cwd=root) + # --long was added in git-1.5.5 + if describe_out is None: + raise NotThisMethod("'git describe' failed") + describe_out = describe_out.strip() + full_out = run_command(GITS, ["rev-parse", "HEAD"], cwd=root) + if full_out is None: + raise NotThisMethod("'git rev-parse' failed") + full_out = full_out.strip() + + pieces = {} + pieces["long"] = full_out + pieces["short"] = full_out[:7] # maybe improved later + pieces["error"] = None + + # parse describe_out. It will be like TAG-NUM-gHEX[-dirty] or HEX[-dirty] + # TAG might have hyphens. + git_describe = describe_out + + # look for -dirty suffix + dirty = git_describe.endswith("-dirty") + pieces["dirty"] = dirty + if dirty: + git_describe = git_describe[:git_describe.rindex("-dirty")] + + # now we have TAG-NUM-gHEX or HEX + + if "-" in git_describe: + # TAG-NUM-gHEX + mo = re.search(r'^(.+)-(\d+)-g([0-9a-f]+)$', git_describe) + if not mo: + # unparseable. Maybe git-describe is misbehaving? + pieces["error"] = ("unable to parse git-describe output: '%%s'" + %% describe_out) + return pieces + + # tag + full_tag = mo.group(1) + if not full_tag.startswith(tag_prefix): + if verbose: + fmt = "tag '%%s' doesn't start with prefix '%%s'" + print(fmt %% (full_tag, tag_prefix)) + pieces["error"] = ("tag '%%s' doesn't start with prefix '%%s'" + %% (full_tag, tag_prefix)) + return pieces + pieces["closest-tag"] = full_tag[len(tag_prefix):] + + # distance: number of commits since tag + pieces["distance"] = int(mo.group(2)) + + # commit: short hex revision ID + pieces["short"] = mo.group(3) + + else: + # HEX: no tags + pieces["closest-tag"] = None + count_out = run_command(GITS, ["rev-list", "HEAD", "--count"], + cwd=root) + pieces["distance"] = int(count_out) # total number of commits + + return pieces + + +def plus_or_dot(pieces): + """Return a + if we don't already have one, else return a .""" + if "+" in pieces.get("closest-tag", ""): + return "." + return "+" + + +def render_pep440(pieces): + """Build up version string, with post-release "local version identifier". + + Our goal: TAG[+DISTANCE.gHEX[.dirty]] . Note that if you + get a tagged build and then dirty it, you'll get TAG+0.gHEX.dirty + + Exceptions: + 1: no tags. git_describe was just HEX. 0+untagged.DISTANCE.gHEX[.dirty] + """ + if pieces["closest-tag"]: + rendered = pieces["closest-tag"] + if pieces["distance"] or pieces["dirty"]: + rendered += plus_or_dot(pieces) + rendered += "%%d.g%%s" %% (pieces["distance"], pieces["short"]) + if pieces["dirty"]: + rendered += ".dirty" + else: + # exception #1 + rendered = "0+untagged.%%d.g%%s" %% (pieces["distance"], + pieces["short"]) + if pieces["dirty"]: + rendered += ".dirty" + return rendered + + +def render_pep440_pre(pieces): + """TAG[.post.devDISTANCE] -- No -dirty. + + Exceptions: + 1: no tags. 0.post.devDISTANCE + """ + if pieces["closest-tag"]: + rendered = pieces["closest-tag"] + if pieces["distance"]: + rendered += ".post.dev%%d" %% pieces["distance"] + else: + # exception #1 + rendered = "0.post.dev%%d" %% pieces["distance"] + return rendered + + +def render_pep440_post(pieces): + """TAG[.postDISTANCE[.dev0]+gHEX] . + + The ".dev0" means dirty. Note that .dev0 sorts backwards + (a dirty tree will appear "older" than the corresponding clean one), + but you shouldn't be releasing software with -dirty anyways. + + Exceptions: + 1: no tags. 0.postDISTANCE[.dev0] + """ + if pieces["closest-tag"]: + rendered = pieces["closest-tag"] + if pieces["distance"] or pieces["dirty"]: + rendered += ".post%%d" %% pieces["distance"] + if pieces["dirty"]: + rendered += ".dev0" + rendered += plus_or_dot(pieces) + rendered += "g%%s" %% pieces["short"] + else: + # exception #1 + rendered = "0.post%%d" %% pieces["distance"] + if pieces["dirty"]: + rendered += ".dev0" + rendered += "+g%%s" %% pieces["short"] + return rendered + + +def render_pep440_old(pieces): + """TAG[.postDISTANCE[.dev0]] . + + The ".dev0" means dirty. + + Eexceptions: + 1: no tags. 0.postDISTANCE[.dev0] + """ + if pieces["closest-tag"]: + rendered = pieces["closest-tag"] + if pieces["distance"] or pieces["dirty"]: + rendered += ".post%%d" %% pieces["distance"] + if pieces["dirty"]: + rendered += ".dev0" + else: + # exception #1 + rendered = "0.post%%d" %% pieces["distance"] + if pieces["dirty"]: + rendered += ".dev0" + return rendered + + +def render_git_describe(pieces): + """TAG[-DISTANCE-gHEX][-dirty]. + + Like 'git describe --tags --dirty --always'. + + Exceptions: + 1: no tags. HEX[-dirty] (note: no 'g' prefix) + """ + if pieces["closest-tag"]: + rendered = pieces["closest-tag"] + if pieces["distance"]: + rendered += "-%%d-g%%s" %% (pieces["distance"], pieces["short"]) + else: + # exception #1 + rendered = pieces["short"] + if pieces["dirty"]: + rendered += "-dirty" + return rendered + + +def render_git_describe_long(pieces): + """TAG-DISTANCE-gHEX[-dirty]. + + Like 'git describe --tags --dirty --always -long'. + The distance/hash is unconditional. + + Exceptions: + 1: no tags. HEX[-dirty] (note: no 'g' prefix) + """ + if pieces["closest-tag"]: + rendered = pieces["closest-tag"] + rendered += "-%%d-g%%s" %% (pieces["distance"], pieces["short"]) + else: + # exception #1 + rendered = pieces["short"] + if pieces["dirty"]: + rendered += "-dirty" + return rendered + + +def render(pieces, style): + """Render the given version pieces into the requested style.""" + if pieces["error"]: + return {"version": "unknown", + "full-revisionid": pieces.get("long"), + "dirty": None, + "error": pieces["error"]} + + if not style or style == "default": + style = "pep440" # the default + + if style == "pep440": + rendered = render_pep440(pieces) + elif style == "pep440-pre": + rendered = render_pep440_pre(pieces) + elif style == "pep440-post": + rendered = render_pep440_post(pieces) + elif style == "pep440-old": + rendered = render_pep440_old(pieces) + elif style == "git-describe": + rendered = render_git_describe(pieces) + elif style == "git-describe-long": + rendered = render_git_describe_long(pieces) + else: + raise ValueError("unknown style '%%s'" %% style) + + return {"version": rendered, "full-revisionid": pieces["long"], + "dirty": pieces["dirty"], "error": None} + + +def get_versions(): + """Get version information or return default if unable to do so.""" + # I am in _version.py, which lives at ROOT/VERSIONFILE_SOURCE. If we have + # __file__, we can work backwards from there to the root. Some + # py2exe/bbfreeze/non-CPython implementations don't do __file__, in which + # case we can only use expanded keywords. + + cfg = get_config() + verbose = cfg.verbose + + try: + return git_versions_from_keywords(get_keywords(), cfg.tag_prefix, + verbose) + except NotThisMethod: + pass + + try: + root = os.path.realpath(__file__) + # versionfile_source is the relative path from the top of the source + # tree (where the .git directory might live) to this file. Invert + # this to find the root from __file__. + for i in cfg.versionfile_source.split('/'): + root = os.path.dirname(root) + except NameError: + return {"version": "0+unknown", "full-revisionid": None, + "dirty": None, + "error": "unable to find root of source tree"} + + try: + pieces = git_pieces_from_vcs(cfg.tag_prefix, root, verbose) + return render(pieces, cfg.style) + except NotThisMethod: + pass + + try: + if cfg.parentdir_prefix: + return versions_from_parentdir(cfg.parentdir_prefix, root, verbose) + except NotThisMethod: + pass + + return {"version": "0+unknown", "full-revisionid": None, + "dirty": None, + "error": "unable to compute version"} +''' + + +@register_vcs_handler("git", "get_keywords") +def git_get_keywords(versionfile_abs): + """Extract version information from the given file.""" + # the code embedded in _version.py can just fetch the value of these + # keywords. When used from setup.py, we don't want to import _version.py, + # so we do it with a regexp instead. This function is not used from + # _version.py. + keywords = {} + try: + f = open(versionfile_abs, "r") + for line in f.readlines(): + if line.strip().startswith("git_refnames ="): + mo = re.search(r'=\s*"(.*)"', line) + if mo: + keywords["refnames"] = mo.group(1) + if line.strip().startswith("git_full ="): + mo = re.search(r'=\s*"(.*)"', line) + if mo: + keywords["full"] = mo.group(1) + f.close() + except EnvironmentError: + pass + return keywords + + +@register_vcs_handler("git", "keywords") +def git_versions_from_keywords(keywords, tag_prefix, verbose): + """Get version information from git keywords.""" + if not keywords: + raise NotThisMethod("no keywords at all, weird") + refnames = keywords["refnames"].strip() + if refnames.startswith("$Format"): + if verbose: + print("keywords are unexpanded, not using") + raise NotThisMethod("unexpanded keywords, not a git-archive tarball") + refs = set([r.strip() for r in refnames.strip("()").split(",")]) + # starting in git-1.8.3, tags are listed as "tag: foo-1.0" instead of + # just "foo-1.0". If we see a "tag: " prefix, prefer those. + TAG = "tag: " + tags = set([r[len(TAG):] for r in refs if r.startswith(TAG)]) + if not tags: + # Either we're using git < 1.8.3, or there really are no tags. We use + # a heuristic: assume all version tags have a digit. The old git %d + # expansion behaves like git log --decorate=short and strips out the + # refs/heads/ and refs/tags/ prefixes that would let us distinguish + # between branches and tags. By ignoring refnames without digits, we + # filter out many common branch names like "release" and + # "stabilization", as well as "HEAD" and "master". + tags = set([r for r in refs if re.search(r'\d', r)]) + if verbose: + print("discarding '%s', no digits" % ",".join(refs-tags)) + if verbose: + print("likely tags: %s" % ",".join(sorted(tags))) + for ref in sorted(tags): + # sorting will prefer e.g. "2.0" over "2.0rc1" + if ref.startswith(tag_prefix): + r = ref[len(tag_prefix):] + if verbose: + print("picking %s" % r) + return {"version": r, + "full-revisionid": keywords["full"].strip(), + "dirty": False, "error": None + } + # no suitable tags, so version is "0+unknown", but full hex is still there + if verbose: + print("no suitable tags, using unknown + full revision id") + return {"version": "0+unknown", + "full-revisionid": keywords["full"].strip(), + "dirty": False, "error": "no suitable tags"} + + +@register_vcs_handler("git", "pieces_from_vcs") +def git_pieces_from_vcs(tag_prefix, root, verbose, run_command=run_command): + """Get version from 'git describe' in the root of the source tree. + + This only gets called if the git-archive 'subst' keywords were *not* + expanded, and _version.py hasn't already been rewritten with a short + version string, meaning we're inside a checked out source tree. + """ + if not os.path.exists(os.path.join(root, ".git")): + if verbose: + print("no .git in %s" % root) + raise NotThisMethod("no .git directory") + + GITS = ["git"] + if sys.platform == "win32": + GITS = ["git.cmd", "git.exe"] + # if there is a tag matching tag_prefix, this yields TAG-NUM-gHEX[-dirty] + # if there isn't one, this yields HEX[-dirty] (no NUM) + describe_out = run_command(GITS, ["describe", "--tags", "--dirty", + "--always", "--long", + "--match", "%s*" % tag_prefix], + cwd=root) + # --long was added in git-1.5.5 + if describe_out is None: + raise NotThisMethod("'git describe' failed") + describe_out = describe_out.strip() + full_out = run_command(GITS, ["rev-parse", "HEAD"], cwd=root) + if full_out is None: + raise NotThisMethod("'git rev-parse' failed") + full_out = full_out.strip() + + pieces = {} + pieces["long"] = full_out + pieces["short"] = full_out[:7] # maybe improved later + pieces["error"] = None + + # parse describe_out. It will be like TAG-NUM-gHEX[-dirty] or HEX[-dirty] + # TAG might have hyphens. + git_describe = describe_out + + # look for -dirty suffix + dirty = git_describe.endswith("-dirty") + pieces["dirty"] = dirty + if dirty: + git_describe = git_describe[:git_describe.rindex("-dirty")] + + # now we have TAG-NUM-gHEX or HEX + + if "-" in git_describe: + # TAG-NUM-gHEX + mo = re.search(r'^(.+)-(\d+)-g([0-9a-f]+)$', git_describe) + if not mo: + # unparseable. Maybe git-describe is misbehaving? + pieces["error"] = ("unable to parse git-describe output: '%s'" + % describe_out) + return pieces + + # tag + full_tag = mo.group(1) + if not full_tag.startswith(tag_prefix): + if verbose: + fmt = "tag '%s' doesn't start with prefix '%s'" + print(fmt % (full_tag, tag_prefix)) + pieces["error"] = ("tag '%s' doesn't start with prefix '%s'" + % (full_tag, tag_prefix)) + return pieces + pieces["closest-tag"] = full_tag[len(tag_prefix):] + + # distance: number of commits since tag + pieces["distance"] = int(mo.group(2)) + + # commit: short hex revision ID + pieces["short"] = mo.group(3) + + else: + # HEX: no tags + pieces["closest-tag"] = None + count_out = run_command(GITS, ["rev-list", "HEAD", "--count"], + cwd=root) + pieces["distance"] = int(count_out) # total number of commits + + return pieces + + +def do_vcs_install(manifest_in, versionfile_source, ipy): + """Git-specific installation logic for Versioneer. + + For Git, this means creating/changing .gitattributes to mark _version.py + for export-time keyword substitution. + """ + GITS = ["git"] + if sys.platform == "win32": + GITS = ["git.cmd", "git.exe"] + files = [manifest_in, versionfile_source] + if ipy: + files.append(ipy) + try: + me = __file__ + if me.endswith(".pyc") or me.endswith(".pyo"): + me = os.path.splitext(me)[0] + ".py" + versioneer_file = os.path.relpath(me) + except NameError: + versioneer_file = "versioneer.py" + files.append(versioneer_file) + present = False + try: + f = open(".gitattributes", "r") + for line in f.readlines(): + if line.strip().startswith(versionfile_source): + if "export-subst" in line.strip().split()[1:]: + present = True + f.close() + except EnvironmentError: + pass + if not present: + f = open(".gitattributes", "a+") + f.write("%s export-subst\n" % versionfile_source) + f.close() + files.append(".gitattributes") + run_command(GITS, ["add", "--"] + files) + + +def versions_from_parentdir(parentdir_prefix, root, verbose): + """Try to determine the version from the parent directory name. + + Source tarballs conventionally unpack into a directory that includes + both the project name and a version string. + """ + dirname = os.path.basename(root) + if not dirname.startswith(parentdir_prefix): + if verbose: + print("guessing rootdir is '%s', but '%s' doesn't start with " + "prefix '%s'" % (root, dirname, parentdir_prefix)) + raise NotThisMethod("rootdir doesn't start with parentdir_prefix") + return {"version": dirname[len(parentdir_prefix):], + "full-revisionid": None, + "dirty": False, "error": None} + +SHORT_VERSION_PY = """ +# This file was generated by 'versioneer.py' (0.16) from +# revision-control system data, or from the parent directory name of an +# unpacked source archive. Distribution tarballs contain a pre-generated copy +# of this file. + +import json +import sys + +version_json = ''' +%s +''' # END VERSION_JSON + + +def get_versions(): + return json.loads(version_json) +""" + + +def versions_from_file(filename): + """Try to determine the version from _version.py if present.""" + try: + with open(filename) as f: + contents = f.read() + except EnvironmentError: + raise NotThisMethod("unable to read _version.py") + mo = re.search(r"version_json = '''\n(.*)''' # END VERSION_JSON", + contents, re.M | re.S) + if not mo: + raise NotThisMethod("no version_json in _version.py") + return json.loads(mo.group(1)) + + +def write_to_version_file(filename, versions): + """Write the given version number to the given _version.py file.""" + os.unlink(filename) + contents = json.dumps(versions, sort_keys=True, + indent=1, separators=(",", ": ")) + with open(filename, "w") as f: + f.write(SHORT_VERSION_PY % contents) + + print("set %s to '%s'" % (filename, versions["version"])) + + +def plus_or_dot(pieces): + """Return a + if we don't already have one, else return a .""" + if "+" in pieces.get("closest-tag", ""): + return "." + return "+" + + +def render_pep440(pieces): + """Build up version string, with post-release "local version identifier". + + Our goal: TAG[+DISTANCE.gHEX[.dirty]] . Note that if you + get a tagged build and then dirty it, you'll get TAG+0.gHEX.dirty + + Exceptions: + 1: no tags. git_describe was just HEX. 0+untagged.DISTANCE.gHEX[.dirty] + """ + if pieces["closest-tag"]: + rendered = pieces["closest-tag"] + if pieces["distance"] or pieces["dirty"]: + rendered += plus_or_dot(pieces) + rendered += "%d.g%s" % (pieces["distance"], pieces["short"]) + if pieces["dirty"]: + rendered += ".dirty" + else: + # exception #1 + rendered = "0+untagged.%d.g%s" % (pieces["distance"], + pieces["short"]) + if pieces["dirty"]: + rendered += ".dirty" + return rendered + + +def render_pep440_pre(pieces): + """TAG[.post.devDISTANCE] -- No -dirty. + + Exceptions: + 1: no tags. 0.post.devDISTANCE + """ + if pieces["closest-tag"]: + rendered = pieces["closest-tag"] + if pieces["distance"]: + rendered += ".post.dev%d" % pieces["distance"] + else: + # exception #1 + rendered = "0.post.dev%d" % pieces["distance"] + return rendered + + +def render_pep440_post(pieces): + """TAG[.postDISTANCE[.dev0]+gHEX] . + + The ".dev0" means dirty. Note that .dev0 sorts backwards + (a dirty tree will appear "older" than the corresponding clean one), + but you shouldn't be releasing software with -dirty anyways. + + Exceptions: + 1: no tags. 0.postDISTANCE[.dev0] + """ + if pieces["closest-tag"]: + rendered = pieces["closest-tag"] + if pieces["distance"] or pieces["dirty"]: + rendered += ".post%d" % pieces["distance"] + if pieces["dirty"]: + rendered += ".dev0" + rendered += plus_or_dot(pieces) + rendered += "g%s" % pieces["short"] + else: + # exception #1 + rendered = "0.post%d" % pieces["distance"] + if pieces["dirty"]: + rendered += ".dev0" + rendered += "+g%s" % pieces["short"] + return rendered + + +def render_pep440_old(pieces): + """TAG[.postDISTANCE[.dev0]] . + + The ".dev0" means dirty. + + Eexceptions: + 1: no tags. 0.postDISTANCE[.dev0] + """ + if pieces["closest-tag"]: + rendered = pieces["closest-tag"] + if pieces["distance"] or pieces["dirty"]: + rendered += ".post%d" % pieces["distance"] + if pieces["dirty"]: + rendered += ".dev0" + else: + # exception #1 + rendered = "0.post%d" % pieces["distance"] + if pieces["dirty"]: + rendered += ".dev0" + return rendered + + +def render_git_describe(pieces): + """TAG[-DISTANCE-gHEX][-dirty]. + + Like 'git describe --tags --dirty --always'. + + Exceptions: + 1: no tags. HEX[-dirty] (note: no 'g' prefix) + """ + if pieces["closest-tag"]: + rendered = pieces["closest-tag"] + if pieces["distance"]: + rendered += "-%d-g%s" % (pieces["distance"], pieces["short"]) + else: + # exception #1 + rendered = pieces["short"] + if pieces["dirty"]: + rendered += "-dirty" + return rendered + + +def render_git_describe_long(pieces): + """TAG-DISTANCE-gHEX[-dirty]. + + Like 'git describe --tags --dirty --always -long'. + The distance/hash is unconditional. + + Exceptions: + 1: no tags. HEX[-dirty] (note: no 'g' prefix) + """ + if pieces["closest-tag"]: + rendered = pieces["closest-tag"] + rendered += "-%d-g%s" % (pieces["distance"], pieces["short"]) + else: + # exception #1 + rendered = pieces["short"] + if pieces["dirty"]: + rendered += "-dirty" + return rendered + + +def render(pieces, style): + """Render the given version pieces into the requested style.""" + if pieces["error"]: + return {"version": "unknown", + "full-revisionid": pieces.get("long"), + "dirty": None, + "error": pieces["error"]} + + if not style or style == "default": + style = "pep440" # the default + + if style == "pep440": + rendered = render_pep440(pieces) + elif style == "pep440-pre": + rendered = render_pep440_pre(pieces) + elif style == "pep440-post": + rendered = render_pep440_post(pieces) + elif style == "pep440-old": + rendered = render_pep440_old(pieces) + elif style == "git-describe": + rendered = render_git_describe(pieces) + elif style == "git-describe-long": + rendered = render_git_describe_long(pieces) + else: + raise ValueError("unknown style '%s'" % style) + + return {"version": rendered, "full-revisionid": pieces["long"], + "dirty": pieces["dirty"], "error": None} + + +class VersioneerBadRootError(Exception): + """The project root directory is unknown or missing key files.""" + + +def get_versions(verbose=False): + """Get the project version from whatever source is available. + + Returns dict with two keys: 'version' and 'full'. + """ + if "versioneer" in sys.modules: + # see the discussion in cmdclass.py:get_cmdclass() + del sys.modules["versioneer"] + + root = get_root() + cfg = get_config_from_root(root) + + assert cfg.VCS is not None, "please set [versioneer]VCS= in setup.cfg" + handlers = HANDLERS.get(cfg.VCS) + assert handlers, "unrecognized VCS '%s'" % cfg.VCS + verbose = verbose or cfg.verbose + assert cfg.versionfile_source is not None, \ + "please set versioneer.versionfile_source" + assert cfg.tag_prefix is not None, "please set versioneer.tag_prefix" + + versionfile_abs = os.path.join(root, cfg.versionfile_source) + + # extract version from first of: _version.py, VCS command (e.g. 'git + # describe'), parentdir. This is meant to work for developers using a + # source checkout, for users of a tarball created by 'setup.py sdist', + # and for users of a tarball/zipball created by 'git archive' or github's + # download-from-tag feature or the equivalent in other VCSes. + + get_keywords_f = handlers.get("get_keywords") + from_keywords_f = handlers.get("keywords") + if get_keywords_f and from_keywords_f: + try: + keywords = get_keywords_f(versionfile_abs) + ver = from_keywords_f(keywords, cfg.tag_prefix, verbose) + if verbose: + print("got version from expanded keyword %s" % ver) + return ver + except NotThisMethod: + pass + + try: + ver = versions_from_file(versionfile_abs) + if verbose: + print("got version from file %s %s" % (versionfile_abs, ver)) + return ver + except NotThisMethod: + pass + + from_vcs_f = handlers.get("pieces_from_vcs") + if from_vcs_f: + try: + pieces = from_vcs_f(cfg.tag_prefix, root, verbose) + ver = render(pieces, cfg.style) + if verbose: + print("got version from VCS %s" % ver) + return ver + except NotThisMethod: + pass + + try: + if cfg.parentdir_prefix: + ver = versions_from_parentdir(cfg.parentdir_prefix, root, verbose) + if verbose: + print("got version from parentdir %s" % ver) + return ver + except NotThisMethod: + pass + + if verbose: + print("unable to compute version") + + return {"version": "0+unknown", "full-revisionid": None, + "dirty": None, "error": "unable to compute version"} + + +def get_version(): + """Get the short version string for this project.""" + return get_versions()["version"] + + +def get_cmdclass(): + """Get the custom setuptools/distutils subclasses used by Versioneer.""" + if "versioneer" in sys.modules: + del sys.modules["versioneer"] + # this fixes the "python setup.py develop" case (also 'install' and + # 'easy_install .'), in which subdependencies of the main project are + # built (using setup.py bdist_egg) in the same python process. Assume + # a main project A and a dependency B, which use different versions + # of Versioneer. A's setup.py imports A's Versioneer, leaving it in + # sys.modules by the time B's setup.py is executed, causing B to run + # with the wrong versioneer. Setuptools wraps the sub-dep builds in a + # sandbox that restores sys.modules to it's pre-build state, so the + # parent is protected against the child's "import versioneer". By + # removing ourselves from sys.modules here, before the child build + # happens, we protect the child from the parent's versioneer too. + # Also see https://github.com/warner/python-versioneer/issues/52 + + cmds = {} + + # we add "version" to both distutils and setuptools + from distutils.core import Command + + class cmd_version(Command): + description = "report generated version string" + user_options = [] + boolean_options = [] + + def initialize_options(self): + pass + + def finalize_options(self): + pass + + def run(self): + vers = get_versions(verbose=True) + print("Version: %s" % vers["version"]) + print(" full-revisionid: %s" % vers.get("full-revisionid")) + print(" dirty: %s" % vers.get("dirty")) + if vers["error"]: + print(" error: %s" % vers["error"]) + cmds["version"] = cmd_version + + # we override "build_py" in both distutils and setuptools + # + # most invocation pathways end up running build_py: + # distutils/build -> build_py + # distutils/install -> distutils/build ->.. + # setuptools/bdist_wheel -> distutils/install ->.. + # setuptools/bdist_egg -> distutils/install_lib -> build_py + # setuptools/install -> bdist_egg ->.. + # setuptools/develop -> ? + + # we override different "build_py" commands for both environments + if "setuptools" in sys.modules: + from setuptools.command.build_py import build_py as _build_py + else: + from distutils.command.build_py import build_py as _build_py + + class cmd_build_py(_build_py): + def run(self): + root = get_root() + cfg = get_config_from_root(root) + versions = get_versions() + _build_py.run(self) + # now locate _version.py in the new build/ directory and replace + # it with an updated value + if cfg.versionfile_build: + target_versionfile = os.path.join(self.build_lib, + cfg.versionfile_build) + print("UPDATING %s" % target_versionfile) + write_to_version_file(target_versionfile, versions) + cmds["build_py"] = cmd_build_py + + if "cx_Freeze" in sys.modules: # cx_freeze enabled? + from cx_Freeze.dist import build_exe as _build_exe + + class cmd_build_exe(_build_exe): + def run(self): + root = get_root() + cfg = get_config_from_root(root) + versions = get_versions() + target_versionfile = cfg.versionfile_source + print("UPDATING %s" % target_versionfile) + write_to_version_file(target_versionfile, versions) + + _build_exe.run(self) + os.unlink(target_versionfile) + with open(cfg.versionfile_source, "w") as f: + LONG = LONG_VERSION_PY[cfg.VCS] + f.write(LONG % + {"DOLLAR": "$", + "STYLE": cfg.style, + "TAG_PREFIX": cfg.tag_prefix, + "PARENTDIR_PREFIX": cfg.parentdir_prefix, + "VERSIONFILE_SOURCE": cfg.versionfile_source, + }) + cmds["build_exe"] = cmd_build_exe + del cmds["build_py"] + + # we override different "sdist" commands for both environments + if "setuptools" in sys.modules: + from setuptools.command.sdist import sdist as _sdist + else: + from distutils.command.sdist import sdist as _sdist + + class cmd_sdist(_sdist): + def run(self): + versions = get_versions() + self._versioneer_generated_versions = versions + # unless we update this, the command will keep using the old + # version + self.distribution.metadata.version = versions["version"] + return _sdist.run(self) + + def make_release_tree(self, base_dir, files): + root = get_root() + cfg = get_config_from_root(root) + _sdist.make_release_tree(self, base_dir, files) + # now locate _version.py in the new base_dir directory + # (remembering that it may be a hardlink) and replace it with an + # updated value + target_versionfile = os.path.join(base_dir, cfg.versionfile_source) + print("UPDATING %s" % target_versionfile) + write_to_version_file(target_versionfile, + self._versioneer_generated_versions) + cmds["sdist"] = cmd_sdist + + return cmds + + +CONFIG_ERROR = """ +setup.cfg is missing the necessary Versioneer configuration. You need +a section like: + + [versioneer] + VCS = git + style = pep440 + versionfile_source = src/myproject/_version.py + versionfile_build = myproject/_version.py + tag_prefix = + parentdir_prefix = myproject- + +You will also need to edit your setup.py to use the results: + + import versioneer + setup(version=versioneer.get_version(), + cmdclass=versioneer.get_cmdclass(), ...) + +Please read the docstring in ./versioneer.py for configuration instructions, +edit setup.cfg, and re-run the installer or 'python versioneer.py setup'. +""" + +SAMPLE_CONFIG = """ +# See the docstring in versioneer.py for instructions. Note that you must +# re-run 'versioneer.py setup' after changing this section, and commit the +# resulting files. + +[versioneer] +#VCS = git +#style = pep440 +#versionfile_source = +#versionfile_build = +#tag_prefix = +#parentdir_prefix = + +""" + +INIT_PY_SNIPPET = """ +from ._version import get_versions +__version__ = get_versions()['version'] +del get_versions +""" + + +def do_setup(): + """Main VCS-independent setup function for installing Versioneer.""" + root = get_root() + try: + cfg = get_config_from_root(root) + except (EnvironmentError, configparser.NoSectionError, + configparser.NoOptionError) as e: + if isinstance(e, (EnvironmentError, configparser.NoSectionError)): + print("Adding sample versioneer config to setup.cfg", + file=sys.stderr) + with open(os.path.join(root, "setup.cfg"), "a") as f: + f.write(SAMPLE_CONFIG) + print(CONFIG_ERROR, file=sys.stderr) + return 1 + + print(" creating %s" % cfg.versionfile_source) + with open(cfg.versionfile_source, "w") as f: + LONG = LONG_VERSION_PY[cfg.VCS] + f.write(LONG % {"DOLLAR": "$", + "STYLE": cfg.style, + "TAG_PREFIX": cfg.tag_prefix, + "PARENTDIR_PREFIX": cfg.parentdir_prefix, + "VERSIONFILE_SOURCE": cfg.versionfile_source, + }) + + ipy = os.path.join(os.path.dirname(cfg.versionfile_source), + "__init__.py") + if os.path.exists(ipy): + try: + with open(ipy, "r") as f: + old = f.read() + except EnvironmentError: + old = "" + if INIT_PY_SNIPPET not in old: + print(" appending to %s" % ipy) + with open(ipy, "a") as f: + f.write(INIT_PY_SNIPPET) + else: + print(" %s unmodified" % ipy) + else: + print(" %s doesn't exist, ok" % ipy) + ipy = None + + # Make sure both the top-level "versioneer.py" and versionfile_source + # (PKG/_version.py, used by runtime code) are in MANIFEST.in, so + # they'll be copied into source distributions. Pip won't be able to + # install the package without this. + manifest_in = os.path.join(root, "MANIFEST.in") + simple_includes = set() + try: + with open(manifest_in, "r") as f: + for line in f: + if line.startswith("include "): + for include in line.split()[1:]: + simple_includes.add(include) + except EnvironmentError: + pass + # That doesn't cover everything MANIFEST.in can do + # (http://docs.python.org/2/distutils/sourcedist.html#commands), so + # it might give some false negatives. Appending redundant 'include' + # lines is safe, though. + if "versioneer.py" not in simple_includes: + print(" appending 'versioneer.py' to MANIFEST.in") + with open(manifest_in, "a") as f: + f.write("include versioneer.py\n") + else: + print(" 'versioneer.py' already in MANIFEST.in") + if cfg.versionfile_source not in simple_includes: + print(" appending versionfile_source ('%s') to MANIFEST.in" % + cfg.versionfile_source) + with open(manifest_in, "a") as f: + f.write("include %s\n" % cfg.versionfile_source) + else: + print(" versionfile_source already in MANIFEST.in") + + # Make VCS-specific changes. For git, this means creating/changing + # .gitattributes to mark _version.py for export-time keyword + # substitution. + do_vcs_install(manifest_in, cfg.versionfile_source, ipy) + return 0 + + +def scan_setup_py(): + """Validate the contents of setup.py against Versioneer's expectations.""" + found = set() + setters = False + errors = 0 + with open("setup.py", "r") as f: + for line in f.readlines(): + if "import versioneer" in line: + found.add("import") + if "versioneer.get_cmdclass()" in line: + found.add("cmdclass") + if "versioneer.get_version()" in line: + found.add("get_version") + if "versioneer.VCS" in line: + setters = True + if "versioneer.versionfile_source" in line: + setters = True + if len(found) != 3: + print("") + print("Your setup.py appears to be missing some important items") + print("(but I might be wrong). Please make sure it has something") + print("roughly like the following:") + print("") + print(" import versioneer") + print(" setup( version=versioneer.get_version(),") + print(" cmdclass=versioneer.get_cmdclass(), ...)") + print("") + errors += 1 + if setters: + print("You should remove lines like 'versioneer.VCS = ' and") + print("'versioneer.versionfile_source = ' . This configuration") + print("now lives in setup.cfg, and should be removed from setup.py") + print("") + errors += 1 + return errors + +if __name__ == "__main__": + cmd = sys.argv[1] + if cmd == "setup": + errors = do_setup() + errors += scan_setup_py() + if errors: + sys.exit(1)