exception handling code revised. gevent, pypy, and threadsim vestiges removed.

devel flag removed.
This commit is contained in:
fawce
2012-07-26 16:22:13 -04:00
parent 1f3a2cc3dc
commit abf9c8efa5
13 changed files with 233 additions and 321 deletions
+69 -63
View File
@@ -14,14 +14,9 @@ from setproctitle import setproctitle
# pyzmq
import zmq
# gevent_zeromq
import gevent_zeromq
# zmq_ctypes
#import zmq_ctypes
from zipline.core.monitor import PARAMETERS
from zipline.utils.gpoll import _Poller as GeventPoller
from zipline.protocol import CONTROL_PROTOCOL, COMPONENT_STATE, \
COMPONENT_FAILURE, CONTROL_FRAME, CONTROL_UNFRAME
@@ -29,6 +24,10 @@ log = logbook.Logger('Component')
from zipline.exceptions import ComponentNoInit
class KillSignal(Exception):
def __init__(self):
pass
class Component(object):
"""
@@ -154,35 +153,13 @@ class Component(object):
def do_work(self):
raise NotImplementedError
def init_zmq(self, flavor):
"""
ZMQ in all flavors. Have it your way.
mp - Distinct contexts | pyzmq
green - Same context | gevent_zeromq
pypy - Same context | zmq_ctypes
"""
if flavor == 'mp':
self.zmq = zmq
self.context = self.zmq.Context()
self.zmq_poller = self.zmq.Poller
# The the process title so you can watch it in top
setproctitle(self.__class__.__name__)
return
if flavor == 'green':
self.zmq = gevent_zeromq.zmq
self.context = self.zmq.Context.instance()
self.zmq_poller = GeventPoller
return
if flavor == 'pypy':
self.zmq = zmq
self.context = self.zmq.Context.instance()
self.zmq_poller = self.zmq.Poller
return
raise Exception("Unknown ZeroMQ Flavor")
def init_zmq(self):
self.zmq = zmq
self.context = self.zmq.Context()
self.zmq_poller = self.zmq.Poller
# The the process title so you can watch it in top
setproctitle(self.__class__.__name__)
return
def _run(self):
"""
@@ -200,7 +177,7 @@ class Component(object):
self.done = False # TODO: use state flag
self.sockets = []
self.init_zmq(self.zmq_flavor)
self.init_zmq()
self.setup_poller()
@@ -209,8 +186,6 @@ class Component(object):
self.signal_ready()
self.lock_ready()
self.wait_ready()
# -----------------------
# YOU SHALL NOT PASS!!!!!
@@ -228,14 +203,17 @@ class Component(object):
try:
self._run()
except Exception as exc:
exc_info = sys.exc_info()
self.signal_exception(exc)
if not isinstance(exc, KillSignal):
self.signal_exception(exc)
else:
# if we get a kill signal, forcibly close all the
# sockets.
# exc_info = sys.exc_info()
# self.relay_exception(exc_info[0], exc_info[1], exc_info[2])
self.teardown_sockets()
# Reraise the exception
raise exc_info[0], exc_info[1], exc_info[2]
finally:
self.shutdown()
self.teardown_sockets()
log.info("Exiting %r" % self)
def working(self):
@@ -311,7 +289,6 @@ class Component(object):
# controller that we're done.
elif event == CONTROL_PROTOCOL.SHUTDOWN:
self.signal_done()
self.shutdown()
# =========
# Hard Kill
@@ -336,7 +313,7 @@ class Component(object):
# Echo back the heartbeat identifier to tell the
# controller that this component is still alive and
# doing work
self.control_out.send(heartbeat_frame)
self.control_out.send(heartbeat_frame, self.zmq.NOBLOCK)
self.last_ping = pre_pong
elif self.last_ping and \
time.time() - self.last_ping > PARAMETERS.MAX_COMPONENT_WAIT:
@@ -353,6 +330,7 @@ class Component(object):
Close all zmq sockets safely. This is universal, no matter where
this is running it will need the sockets closed.
"""
log.warn("{id} closing all sockets".format(id=self.get_id))
#close all the sockets
for sock in self.sockets:
sock.close()
@@ -364,6 +342,7 @@ class Component(object):
Tear down after normal operation.
"""
if self.on_done:
log.warn("{id} calling done.".format(id=self.get_id))
self.on_done()
def kill(self):
@@ -373,7 +352,8 @@ class Component(object):
Tear down ( fast ) as a mode of failure in the simulation or on
service halt.
"""
sys.exit(1)
# sys.exit(1)
raise KillSignal()
# ----------------------
# Internal Maintenance
@@ -404,7 +384,7 @@ class Component(object):
start_wait = time.time()
while self.waiting:
socks = dict(self.poll.poll(100))
socks = dict(self.poll.poll(0))
assert self.control_in, \
'Component does not have a control_in socket'
@@ -444,9 +424,7 @@ class Component(object):
# data that are done during a clean shutdown. Inform the
# controller that we're done.
elif event == CONTROL_PROTOCOL.SHUTDOWN:
self.signal_done()
self.shutdown()
break
# =========
@@ -492,11 +470,13 @@ class Component(object):
def signal_exception(self, exc=None, scope=None):
"""
This is a *very* important error tracking handler.
All exceptions inside any component should boil back to
this handler.
Will inform the system that the component has failed and how it
has failed.
"""
if scope == 'algo':
self.error_state = COMPONENT_FAILURE.ALGOEXCEPT
else:
@@ -512,19 +492,47 @@ class Component(object):
exc_type, exc_value, exc_traceback = sys.exc_info()
trace = ''.join(traceback.format_exception(exc_type, exc_value, exc_traceback))
sys.stdout.write(trace)
# if a downstream component fails, this component may try
# sending when there are zero connections to the socket,
# which will raise ZMQError(EAGAIN). So, it doesn't make
# sense to relay this exception to Monitor and the rest
# of the zipline.
if isinstance(exc, zmq.ZMQError) and exc.errno == zmq.EAGAIN:
log.warn("{id} raised a ZMQError(EAGAIN) not relaying"\
.format(id=self.get_id))
return
if hasattr(self, 'exception_callback') and self.exception_callback:
self.exception_callback(exc_type, exc_value, exc_traceback)
# sys.stdout.write(trace)
log.exception("Unexpected error in run for {id}.".format(id=self.get_id))
self.relay_exception(exc_type, exc_value, exc_traceback)
if hasattr(self, 'control_out') and self.control_out:
exception_frame = CONTROL_FRAME(
CONTROL_PROTOCOL.EXCEPTION,
trace
)
self.control_out.send(exception_frame)
try:
log.info('{id} sending exception to controller'.format(id=self.get_id))
exception_frame = CONTROL_FRAME(
CONTROL_PROTOCOL.EXCEPTION,
trace
)
self.control_out.send(exception_frame, self.zmq.NOBLOCK)
# The controller should relay the exception back
# to all zipline components. Wait here until the
# notice arrives, and we can assume other zipline
# components have broken out of their message
# loops.
for i in xrange(100):
self.heartbeat(timeout=1000)
log.warn("{id} Never heard back from monitor."\
.format(id=self.get_id))
except:
log.exception("Exception waiting for controller reply")
def relay_exception(self, exc_type, exc_value, exc_traceback):
if hasattr(self, 'exception_callback') and self.exception_callback:
log.info('{id} making exception callback'.format(id=self.get_id))
self.exception_callback(exc_type, exc_value, exc_traceback)
#LOGGER.exception("Unexpected error in run for {id}.".format(id=self.get_id))
def signal_done(self):
"""
@@ -532,6 +540,8 @@ class Component(object):
"""
self.state_flag = COMPONENT_STATE.DONE
# notify internal work loop that we're done
self.done = True # TODO: use state flag
if hasattr(self, 'out_socket') and self.out_socket:
msg = zmq.Message(str(CONTROL_PROTOCOL.DONE))
@@ -554,8 +564,7 @@ class Component(object):
# last heartbeat, and wait an unusually long time.
self.heartbeat(timeout=5000)
# notify internal work look that we're done
self.done = True # TODO: use state flag
# -----------
@@ -567,9 +576,6 @@ class Component(object):
Setup the poller used for multiplexing the incoming data
handling sockets.
"""
# Initializes the poller class specified by the flavor of
# ZeroMQ. Either zmq.Poller or gpoll.Poller .
self.poll = self.zmq_poller()
def bind_data(self):