mirror of
https://github.com/wassname/ray.git
synced 2026-07-27 11:26:41 +08:00
Experimental: enable automatic GCS flushing with configurable policy. (#2266)
* build_credis.sh: use an up-to-date credis commit. * build_credis.sh: leveldb is updated, so update build cmds for it * WIP: make monitor.py issue flush; switch gcs client to use credis * Experimental: enable automatic GCS flushing with configurable policy. * Fix linux compilation error * Fix leveldb build * Use optimized build for credis * Address comments * Attempt to fix tests
This commit is contained in:
committed by
Philipp Moritz
parent
60bc3a014f
commit
8190ff1fd0
@@ -7,6 +7,8 @@ from .features import (
|
||||
flush_redis_unsafe, flush_task_and_object_metadata_unsafe,
|
||||
flush_finished_tasks_unsafe, flush_evicted_objects_unsafe,
|
||||
_flush_finished_tasks_unsafe_shard, _flush_evicted_objects_unsafe_shard)
|
||||
from .gcs_flush_policy import (set_flushing_policy, GcsFlushPolicy,
|
||||
SimpleGcsFlushPolicy)
|
||||
from .named_actors import get_actor, register_actor
|
||||
from .api import get, wait
|
||||
|
||||
@@ -15,5 +17,6 @@ __all__ = [
|
||||
"flush_task_and_object_metadata_unsafe", "flush_finished_tasks_unsafe",
|
||||
"flush_evicted_objects_unsafe", "_flush_finished_tasks_unsafe_shard",
|
||||
"_flush_evicted_objects_unsafe_shard", "get_actor", "register_actor",
|
||||
"get", "wait"
|
||||
"get", "wait", "set_flushing_policy", "GcsFlushPolicy",
|
||||
"SimpleGcsFlushPolicy"
|
||||
]
|
||||
|
||||
@@ -10,7 +10,7 @@ OBJECT_LOCATION_PREFIX = b"OL:"
|
||||
TASK_PREFIX = b"TT:"
|
||||
|
||||
|
||||
def flush_redis_unsafe():
|
||||
def flush_redis_unsafe(redis_client=None):
|
||||
"""This removes some non-critical state from the primary Redis shard.
|
||||
|
||||
This removes the log files as well as the event log from Redis. This can
|
||||
@@ -18,10 +18,14 @@ def flush_redis_unsafe():
|
||||
of metadata in Redis. However, it will only partially address the issue as
|
||||
much of the data is in the task table (and object table), which are not
|
||||
flushed.
|
||||
"""
|
||||
ray.worker.global_worker.check_connected()
|
||||
|
||||
redis_client = ray.worker.global_worker.redis_client
|
||||
Args:
|
||||
redis_client: optional, if not provided then ray.init() must have been
|
||||
called.
|
||||
"""
|
||||
if redis_client is None:
|
||||
ray.worker.global_worker.check_connected()
|
||||
redis_client = ray.worker.global_worker.redis_client
|
||||
|
||||
# Delete the log files from the primary Redis shard.
|
||||
keys = redis_client.keys("LOGFILE:*")
|
||||
|
||||
@@ -0,0 +1,91 @@
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import os
|
||||
import time
|
||||
|
||||
import ray
|
||||
import ray.cloudpickle as pickle
|
||||
|
||||
|
||||
class GcsFlushPolicy(object):
|
||||
"""Experimental: a policy to control GCS flushing.
|
||||
|
||||
Used by Monitor to enable automatic control of memory usage.
|
||||
"""
|
||||
|
||||
def should_flush(self, redis_client):
|
||||
"""Returns a bool, whether a flush request should be issued."""
|
||||
pass
|
||||
|
||||
def num_entries_to_flush(self):
|
||||
"""Returns an upper bound for number of entries to flush next."""
|
||||
pass
|
||||
|
||||
def record_flush(self):
|
||||
"""Must be called after a flush has been performed."""
|
||||
pass
|
||||
|
||||
|
||||
class SimpleGcsFlushPolicy(GcsFlushPolicy):
|
||||
"""A simple policy with constant flush rate, after a warmup period.
|
||||
|
||||
Example policy values:
|
||||
flush_when_at_least_bytes 2GB
|
||||
flush_period_secs 10s
|
||||
flush_num_entries_each_time 10k
|
||||
|
||||
This means
|
||||
(1) If the GCS shard uses less than 2GB of memory, no flushing would take
|
||||
place. This should cover most Ray runs.
|
||||
(2) The GCS shard will only honor a flush request, if it's issued after 10
|
||||
seconds since the last processed flush. In particular this means it's
|
||||
okay for the Monitor to issue requests more frequently than this param.
|
||||
(3) When processing a flush, the shard will flush at most 10k entries.
|
||||
This is to control the latency of each request.
|
||||
|
||||
Note, flush rate == (flush period) * (num entries each time). So
|
||||
applications that have a heavier GCS load can tune these params.
|
||||
"""
|
||||
|
||||
def __init__(self,
|
||||
flush_when_at_least_bytes=(1 << 31),
|
||||
flush_period_secs=10,
|
||||
flush_num_entries_each_time=10000):
|
||||
self.flush_when_at_least_bytes = flush_when_at_least_bytes
|
||||
self.flush_period_secs = flush_period_secs
|
||||
self.flush_num_entries_each_time = flush_num_entries_each_time
|
||||
self.last_flush_timestamp = time.time()
|
||||
|
||||
def should_flush(self, redis_client):
|
||||
if time.time() - self.last_flush_timestamp < self.flush_period_secs:
|
||||
return False
|
||||
|
||||
used_memory = redis_client.info("memory")["used_memory"]
|
||||
assert used_memory > 0
|
||||
|
||||
return used_memory >= self.flush_when_at_least_bytes
|
||||
|
||||
def num_entries_to_flush(self):
|
||||
return self.flush_num_entries_each_time
|
||||
|
||||
def record_flush(self):
|
||||
self.last_flush_timestamp = time.time()
|
||||
|
||||
def serialize(self):
|
||||
return pickle.dumps(self)
|
||||
|
||||
|
||||
def set_flushing_policy(flushing_policy):
|
||||
"""Serialize this policy for Monitor to pick up."""
|
||||
if "RAY_USE_NEW_GCS" not in os.environ:
|
||||
raise Exception(
|
||||
"set_flushing_policy() is only available when environment "
|
||||
"variable RAY_USE_NEW_GCS is present at both compile and run time."
|
||||
)
|
||||
ray.worker.global_worker.check_connected()
|
||||
redis_client = ray.worker.global_worker.redis_client
|
||||
|
||||
serialized = pickle.dumps(flushing_policy)
|
||||
redis_client.set("gcs_flushing_policy", serialized)
|
||||
Reference in New Issue
Block a user