mirror of
https://github.com/wassname/ray.git
synced 2026-08-11 11:24:51 +08:00
Lint Python files with Yapf (#1872)
This commit is contained in:
committed by
Robert Nishihara
parent
a3ddde398c
commit
74162d1492
@@ -11,7 +11,6 @@ import pyarrow as pa
|
||||
|
||||
|
||||
class ComponentFailureTest(unittest.TestCase):
|
||||
|
||||
def tearDown(self):
|
||||
ray.worker.cleanup()
|
||||
|
||||
@@ -24,54 +23,20 @@ class ComponentFailureTest(unittest.TestCase):
|
||||
def f():
|
||||
ray.worker.global_worker.plasma_client.get(obj_id)
|
||||
|
||||
ray.worker._init(num_workers=1,
|
||||
driver_mode=ray.SILENT_MODE,
|
||||
start_workers_from_local_scheduler=False,
|
||||
start_ray_local=True,
|
||||
redirect_output=True)
|
||||
ray.worker._init(
|
||||
num_workers=1,
|
||||
driver_mode=ray.SILENT_MODE,
|
||||
start_workers_from_local_scheduler=False,
|
||||
start_ray_local=True,
|
||||
redirect_output=True)
|
||||
|
||||
# Have the worker wait in a get call.
|
||||
f.remote()
|
||||
|
||||
# Kill the worker.
|
||||
time.sleep(1)
|
||||
(ray.services
|
||||
.all_processes[ray.services.PROCESS_TYPE_WORKER][0].terminate())
|
||||
time.sleep(0.1)
|
||||
|
||||
# Seal the object so the store attempts to notify the worker that the
|
||||
# get has been fulfilled.
|
||||
ray.worker.global_worker.plasma_client.create(
|
||||
pa.plasma.ObjectID(obj_id), 100)
|
||||
ray.worker.global_worker.plasma_client.seal(pa.plasma.ObjectID(obj_id))
|
||||
time.sleep(0.1)
|
||||
|
||||
# Make sure that nothing has died.
|
||||
self.assertTrue(ray.services.all_processes_alive(
|
||||
exclude=[ray.services.PROCESS_TYPE_WORKER]))
|
||||
|
||||
# This test checks that when a worker dies in the middle of a wait, the
|
||||
# plasma store and manager will not die.
|
||||
def testDyingWorkerWait(self):
|
||||
obj_id = 20 * b"a"
|
||||
|
||||
@ray.remote
|
||||
def f():
|
||||
ray.worker.global_worker.plasma_client.wait([obj_id])
|
||||
|
||||
ray.worker._init(num_workers=1,
|
||||
driver_mode=ray.SILENT_MODE,
|
||||
start_workers_from_local_scheduler=False,
|
||||
start_ray_local=True,
|
||||
redirect_output=True)
|
||||
|
||||
# Have the worker wait in a get call.
|
||||
f.remote()
|
||||
|
||||
# Kill the worker.
|
||||
time.sleep(1)
|
||||
(ray.services
|
||||
.all_processes[ray.services.PROCESS_TYPE_WORKER][0].terminate())
|
||||
(ray.services.all_processes[ray.services.PROCESS_TYPE_WORKER][0]
|
||||
.terminate())
|
||||
time.sleep(0.1)
|
||||
|
||||
# Seal the object so the store attempts to notify the worker that the
|
||||
@@ -82,8 +47,46 @@ class ComponentFailureTest(unittest.TestCase):
|
||||
time.sleep(0.1)
|
||||
|
||||
# Make sure that nothing has died.
|
||||
self.assertTrue(ray.services.all_processes_alive(
|
||||
exclude=[ray.services.PROCESS_TYPE_WORKER]))
|
||||
self.assertTrue(
|
||||
ray.services.all_processes_alive(
|
||||
exclude=[ray.services.PROCESS_TYPE_WORKER]))
|
||||
|
||||
# This test checks that when a worker dies in the middle of a wait, the
|
||||
# plasma store and manager will not die.
|
||||
def testDyingWorkerWait(self):
|
||||
obj_id = 20 * b"a"
|
||||
|
||||
@ray.remote
|
||||
def f():
|
||||
ray.worker.global_worker.plasma_client.wait([obj_id])
|
||||
|
||||
ray.worker._init(
|
||||
num_workers=1,
|
||||
driver_mode=ray.SILENT_MODE,
|
||||
start_workers_from_local_scheduler=False,
|
||||
start_ray_local=True,
|
||||
redirect_output=True)
|
||||
|
||||
# Have the worker wait in a get call.
|
||||
f.remote()
|
||||
|
||||
# Kill the worker.
|
||||
time.sleep(1)
|
||||
(ray.services.all_processes[ray.services.PROCESS_TYPE_WORKER][0]
|
||||
.terminate())
|
||||
time.sleep(0.1)
|
||||
|
||||
# Seal the object so the store attempts to notify the worker that the
|
||||
# get has been fulfilled.
|
||||
ray.worker.global_worker.plasma_client.create(
|
||||
pa.plasma.ObjectID(obj_id), 100)
|
||||
ray.worker.global_worker.plasma_client.seal(pa.plasma.ObjectID(obj_id))
|
||||
time.sleep(0.1)
|
||||
|
||||
# Make sure that nothing has died.
|
||||
self.assertTrue(
|
||||
ray.services.all_processes_alive(
|
||||
exclude=[ray.services.PROCESS_TYPE_WORKER]))
|
||||
|
||||
def _testWorkerFailed(self, num_local_schedulers):
|
||||
@ray.remote
|
||||
@@ -92,23 +95,25 @@ class ComponentFailureTest(unittest.TestCase):
|
||||
return x
|
||||
|
||||
num_initial_workers = 4
|
||||
ray.worker._init(num_workers=(num_initial_workers *
|
||||
num_local_schedulers),
|
||||
num_local_schedulers=num_local_schedulers,
|
||||
start_workers_from_local_scheduler=False,
|
||||
start_ray_local=True,
|
||||
num_cpus=[num_initial_workers] * num_local_schedulers,
|
||||
redirect_output=True)
|
||||
ray.worker._init(
|
||||
num_workers=(num_initial_workers * num_local_schedulers),
|
||||
num_local_schedulers=num_local_schedulers,
|
||||
start_workers_from_local_scheduler=False,
|
||||
start_ray_local=True,
|
||||
num_cpus=[num_initial_workers] * num_local_schedulers,
|
||||
redirect_output=True)
|
||||
# Submit more tasks than there are workers so that all workers and
|
||||
# cores are utilized.
|
||||
object_ids = [f.remote(i) for i
|
||||
in range(num_initial_workers * num_local_schedulers)]
|
||||
object_ids = [
|
||||
f.remote(i)
|
||||
for i in range(num_initial_workers * num_local_schedulers)
|
||||
]
|
||||
object_ids += [f.remote(object_id) for object_id in object_ids]
|
||||
# Allow the tasks some time to begin executing.
|
||||
time.sleep(0.1)
|
||||
# Kill the workers as the tasks execute.
|
||||
for worker in (ray.services
|
||||
.all_processes[ray.services.PROCESS_TYPE_WORKER]):
|
||||
for worker in (
|
||||
ray.services.all_processes[ray.services.PROCESS_TYPE_WORKER]):
|
||||
worker.terminate()
|
||||
time.sleep(0.1)
|
||||
# Make sure that we can still get the objects after the executing tasks
|
||||
@@ -123,6 +128,7 @@ class ComponentFailureTest(unittest.TestCase):
|
||||
|
||||
def _testComponentFailed(self, component_type):
|
||||
"""Kill a component on all worker nodes and check workload succeeds."""
|
||||
|
||||
@ray.remote
|
||||
def f(x, j):
|
||||
time.sleep(0.2)
|
||||
@@ -140,9 +146,10 @@ class ComponentFailureTest(unittest.TestCase):
|
||||
|
||||
# Submit more tasks than there are workers so that all workers and
|
||||
# cores are utilized.
|
||||
object_ids = [f.remote(i, 0) for i
|
||||
in range(num_workers_per_scheduler *
|
||||
num_local_schedulers)]
|
||||
object_ids = [
|
||||
f.remote(i, 0)
|
||||
for i in range(num_workers_per_scheduler * num_local_schedulers)
|
||||
]
|
||||
object_ids += [f.remote(object_id, 1) for object_id in object_ids]
|
||||
object_ids += [f.remote(object_id, 2) for object_id in object_ids]
|
||||
|
||||
@@ -162,8 +169,8 @@ class ComponentFailureTest(unittest.TestCase):
|
||||
# Make sure that we can still get the objects after the executing tasks
|
||||
# died.
|
||||
results = ray.get(object_ids)
|
||||
expected_results = 4 * list(range(
|
||||
num_workers_per_scheduler * num_local_schedulers))
|
||||
expected_results = 4 * list(
|
||||
range(num_workers_per_scheduler * num_local_schedulers))
|
||||
self.assertEqual(results, expected_results)
|
||||
|
||||
def check_components_alive(self, component_type, check_component_alive):
|
||||
@@ -182,8 +189,7 @@ class ComponentFailureTest(unittest.TestCase):
|
||||
self.assertTrue(not component.poll() is None)
|
||||
|
||||
@unittest.skipIf(
|
||||
os.environ.get('RAY_USE_NEW_GCS', False),
|
||||
"Hanging with new GCS API.")
|
||||
os.environ.get('RAY_USE_NEW_GCS', False), "Hanging with new GCS API.")
|
||||
def testLocalSchedulerFailed(self):
|
||||
# Kill all local schedulers on worker nodes.
|
||||
self._testComponentFailed(ray.services.PROCESS_TYPE_LOCAL_SCHEDULER)
|
||||
@@ -198,8 +204,7 @@ class ComponentFailureTest(unittest.TestCase):
|
||||
False)
|
||||
|
||||
@unittest.skipIf(
|
||||
os.environ.get('RAY_USE_NEW_GCS', False),
|
||||
"Hanging with new GCS API.")
|
||||
os.environ.get('RAY_USE_NEW_GCS', False), "Hanging with new GCS API.")
|
||||
def testPlasmaManagerFailed(self):
|
||||
# Kill all plasma managers on worker nodes.
|
||||
self._testComponentFailed(ray.services.PROCESS_TYPE_PLASMA_MANAGER)
|
||||
@@ -214,8 +219,7 @@ class ComponentFailureTest(unittest.TestCase):
|
||||
False)
|
||||
|
||||
@unittest.skipIf(
|
||||
os.environ.get('RAY_USE_NEW_GCS', False),
|
||||
"Hanging with new GCS API.")
|
||||
os.environ.get('RAY_USE_NEW_GCS', False), "Hanging with new GCS API.")
|
||||
def testPlasmaStoreFailed(self):
|
||||
# Kill all plasma stores on worker nodes.
|
||||
self._testComponentFailed(ray.services.PROCESS_TYPE_PLASMA_STORE)
|
||||
@@ -235,7 +239,8 @@ class ComponentFailureTest(unittest.TestCase):
|
||||
all_processes[ray.services.PROCESS_TYPE_PLASMA_STORE][0],
|
||||
all_processes[ray.services.PROCESS_TYPE_PLASMA_MANAGER][0],
|
||||
all_processes[ray.services.PROCESS_TYPE_LOCAL_SCHEDULER][0],
|
||||
all_processes[ray.services.PROCESS_TYPE_GLOBAL_SCHEDULER][0]]
|
||||
all_processes[ray.services.PROCESS_TYPE_GLOBAL_SCHEDULER][0]
|
||||
]
|
||||
|
||||
# Kill all the components sequentially.
|
||||
for process in processes:
|
||||
@@ -253,7 +258,8 @@ class ComponentFailureTest(unittest.TestCase):
|
||||
all_processes[ray.services.PROCESS_TYPE_PLASMA_STORE][0],
|
||||
all_processes[ray.services.PROCESS_TYPE_PLASMA_MANAGER][0],
|
||||
all_processes[ray.services.PROCESS_TYPE_LOCAL_SCHEDULER][0],
|
||||
all_processes[ray.services.PROCESS_TYPE_GLOBAL_SCHEDULER][0]]
|
||||
all_processes[ray.services.PROCESS_TYPE_GLOBAL_SCHEDULER][0]
|
||||
]
|
||||
|
||||
# Kill all the components in parallel.
|
||||
for process in processes:
|
||||
|
||||
Reference in New Issue
Block a user