mirror of
https://github.com/wassname/catalyst.git
synced 2026-07-29 11:18:20 +08:00
Adds the data bundle concept which makes it easy for users to register loading functions to build out minute and daily data along with an assets db and adjustments db. By default we have provided a `quandl` bundle which pulls from the public domain WIKI dataset. Users may register new bundles by decorating an ingest function with `zipline.data.bundles.register(<name>)`. This also provides a `yahoo_equities` function for creating an ingestion function that will load a static set of assets from yahoo. The cli is now structured as a couple of subcommands and has been changed to `python -m zipline`. The old behavior of `run_algo.py` has been moved to the `run` subcommand. This is almost entirely the same except that it now takes the name of the data bundle to use, defaulting to `quandl`. The next subcommand is `ingest` which takes the name of a data bundle to ingest. This will run the loading machinery and write the data to a specified location that `run` can find. There is also a `clean` subcommand which deletes the data that was written with `ingest`. Extensions have also been added to zipline. This is an experimental feature where users can provide an extra set of python files to run at the start of the process. These can be used to configure aspects of zipline. Right now the only thing that is supported in an extension file is the registration of a new data bundle.
165 lines
5.1 KiB
Python
165 lines
5.1 KiB
Python
import os
|
|
|
|
import numpy as np
|
|
import pandas as pd
|
|
from pandas_datareader.data import DataReader
|
|
import requests
|
|
|
|
from zipline.utils.cli import maybe_show_progress
|
|
|
|
|
|
def _cachpath(symbol, type_):
|
|
return '-'.join((symbol.replace(os.path.sep, '_'), type_))
|
|
|
|
|
|
def yahoo_equities(symbols, start=None, end=None):
|
|
"""Create a data bundle ingest function from a set of symbols loaded from
|
|
yahoo.
|
|
|
|
Parameters
|
|
----------
|
|
symbols : iterable[str]
|
|
The ticker symbols to load data for.
|
|
start : datetime, optional
|
|
The start date to query for. By default this pulls the full history
|
|
for the calendar.
|
|
end : datetime, optional
|
|
The end date to query for. By default this pulls the full history
|
|
for the calendar.
|
|
|
|
Returns
|
|
-------
|
|
ingest : callable
|
|
The bundle ingest function for the given set of symbols.
|
|
|
|
Examples
|
|
--------
|
|
This code should be added to ~/.zipline/extension.py
|
|
|
|
.. code-block:: python
|
|
|
|
from zipline.data.bundles import yahoo_equities, register
|
|
|
|
symbols = (
|
|
'AAPL',
|
|
'IBM',
|
|
'MSFT',
|
|
)
|
|
register('my_bundle', yahoo_equities(symbols))
|
|
|
|
Notes
|
|
-----
|
|
The sids for each symbol will be the index into the symbols sequence.
|
|
"""
|
|
# strict this in memory so that we can reiterate over it
|
|
symbols = tuple(symbols)
|
|
|
|
def ingest(environ,
|
|
asset_db_writer,
|
|
minute_bar_writer, # unused
|
|
daily_bar_writer,
|
|
adjustment_writer,
|
|
calendar,
|
|
cache,
|
|
show_progress,
|
|
# pass these as defaults to make them 'nonlocal' in py2
|
|
start=start,
|
|
end=end):
|
|
if start is None:
|
|
start = calendar[0]
|
|
if end is None:
|
|
end = None
|
|
|
|
metadata = pd.DataFrame(np.empty(len(symbols), dtype=[
|
|
('start_date', 'datetime64[ns]'),
|
|
('end_date', 'datetime64[ns]'),
|
|
('symbol', 'object'),
|
|
]))
|
|
|
|
def _pricing_iter():
|
|
sid = 0
|
|
with maybe_show_progress(
|
|
symbols,
|
|
show_progress,
|
|
label='Downloading Yahoo pricing data: ') as it, \
|
|
requests.Session() as session:
|
|
for symbol in it:
|
|
path = _cachpath(symbol, 'ohlcv')
|
|
try:
|
|
df = cache[path]
|
|
except KeyError:
|
|
df = cache[path] = DataReader(
|
|
symbol,
|
|
'yahoo',
|
|
start,
|
|
end,
|
|
session=session,
|
|
).sort_index()
|
|
|
|
# the start date is the date of the first trade and
|
|
# the end date is the date of the last trade
|
|
metadata.iloc[sid] = df.index[0], df.index[-1], symbol
|
|
df.rename(
|
|
columns={
|
|
'Open': 'open',
|
|
'High': 'high',
|
|
'Low': 'low',
|
|
'Close': 'close',
|
|
'Volume': 'volume',
|
|
},
|
|
inplace=True,
|
|
)
|
|
yield sid, df
|
|
sid += 1
|
|
|
|
daily_bar_writer.write(_pricing_iter(), show_progress=True)
|
|
|
|
symbol_map = pd.Series(metadata.symbol.index, metadata.symbol)
|
|
asset_db_writer.write(equities=metadata)
|
|
|
|
adjustments = []
|
|
with maybe_show_progress(
|
|
symbols,
|
|
show_progress,
|
|
label='Downloading Yahoo adjustment data: ') as it, \
|
|
requests.Session() as session:
|
|
for symbol in it:
|
|
path = _cachpath(symbol, 'adjustment')
|
|
try:
|
|
df = cache[path]
|
|
except KeyError:
|
|
df = cache[path] = DataReader(
|
|
symbol,
|
|
'yahoo-actions',
|
|
start,
|
|
end,
|
|
session=session,
|
|
).sort_index()
|
|
|
|
df['sid'] = symbol_map[symbol]
|
|
adjustments.append(df)
|
|
|
|
adj_df = pd.concat(adjustments)
|
|
adj_df.index.name = 'date'
|
|
adj_df.reset_index(inplace=True)
|
|
|
|
splits = adj_df[adj_df.action == 'SPLIT']
|
|
splits = splits.rename(
|
|
columns={'value': 'ratio', 'date': 'effective_date'},
|
|
)
|
|
splits.drop('action', axis=1, inplace=True)
|
|
|
|
dividends = adj_df[adj_df.action == 'DIVIDEND']
|
|
dividends = dividends.rename(
|
|
columns={'value': 'amount', 'date': 'ex_date'},
|
|
)
|
|
dividends.drop('action', axis=1, inplace=True)
|
|
# we do not have this data in the yahoo dataset
|
|
dividends['record_date'] = pd.NaT
|
|
dividends['declared_date'] = pd.NaT
|
|
dividends['pay_date'] = pd.NaT
|
|
|
|
adjustment_writer.write(splits=splits, dividends=dividends)
|
|
|
|
return ingest
|