mirror of
https://github.com/wassname/ray.git
synced 2026-10-01 12:31:37 +08:00
[rllib] RLlib in 60 seconds documentation (#5430)
This commit is contained in:
1 parent
3218ee389a
commit
79949fb8a0
5 files changed
+274
-180
No files matched your search
@@ -200,6 +200,7 @@ The following are good places to discuss Ray.
|
||||
:caption: RLlib
|
||||
|
||||
rllib.rst
|
||||
rllib-toc.rst
|
||||
rllib-training.rst
|
||||
rllib-env.rst
|
||||
rllib-models.rst
|
||||
|
||||
@@ -0,0 +1 @@
|
||||
<svg version="1.1" viewBox="0.0 0.0 844.6797900262467 100.0" fill="none" stroke="none" stroke-linecap="square" stroke-miterlimit="10" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns="http://www.w3.org/2000/svg"><clipPath id="p.0"><path d="m0 0l844.6798 0l0 100.0l-844.6798 0l0 -100.0z" clip-rule="nonzero"/></clipPath><g clip-path="url(#p.0)"><path fill="#000000" fill-opacity="0.0" d="m0 0l844.6798 0l0 100.0l-844.6798 0z" fill-rule="evenodd"/><path fill="#ffffff" d="m4.4409447 13.635171l134.07874 0l0 70.42519l-134.07874 0z" fill-rule="evenodd"/><path stroke="#000000" stroke-width="1.0" stroke-linejoin="round" stroke-linecap="butt" d="m4.4409447 13.635171l134.07874 0l0 70.42519l-134.07874 0z" fill-rule="evenodd"/><path fill="#000000" d="m14.36282 56.564644l1.59375 0.234375q0.109375 0.75 0.5625 1.078125q0.609375 0.453125 1.671875 0.453125q1.140625 0 1.75 -0.453125q0.625 -0.453125 0.84375 -1.265625q0.125 -0.5 0.109375 -2.109375q-1.0625 1.265625 -2.671875 1.265625q-2.0 0 -3.09375 -1.4375q-1.09375 -1.4375 -1.09375 -3.453125q0 -1.390625 0.5 -2.5625q0.515625 -1.171875 1.453125 -1.796875q0.953125 -0.640625 2.25 -0.640625q1.703125 0 2.8125 1.375l0 -1.15625l1.515625 0l0 8.359375q0 2.265625 -0.46875 3.203125q-0.453125 0.9375 -1.453125 1.484375q-0.984375 0.546875 -2.453125 0.546875q-1.71875 0 -2.796875 -0.78125q-1.0625 -0.765625 -1.03125 -2.34375zm1.359375 -5.8125q0 1.90625 0.75 2.78125q0.765625 0.875 1.90625 0.875q1.125 0 1.890625 -0.859375q0.765625 -0.875 0.765625 -2.734375q0 -1.78125 -0.796875 -2.671875q-0.78125 -0.90625 -1.890625 -0.90625q-1.09375 0 -1.859375 0.890625q-0.765625 0.875 -0.765625 2.625zm9.250717 8.734375l-0.1875 -1.53125q0.546875 0.140625 0.9375 0.140625q0.546875 0 0.875 -0.1875q0.328125 -0.171875 0.546875 -0.5q0.15625 -0.25 0.5 -1.21875q0.046875 -0.140625 0.140625 -0.40625l-3.671875 -9.6875l1.765625 0l2.015625 5.59375q0.390625 1.078125 0.703125 2.25q0.28125 -1.125 0.671875 -2.203125l2.078125 -5.640625l1.640625 0l-3.6875 9.828125q-0.59375 1.609375 -0.921875 2.203125q-0.4375 0.8125 -1.0 1.1875q-0.5625 0.375 -1.34375 0.375q-0.484375 0 -1.0625 -0.203125zm9.40625 -3.71875l0 -9.671875l1.46875 0l0 1.359375q0.453125 -0.71875 1.203125 -1.140625q0.765625 -0.4375 1.71875 -0.4375q1.078125 0 1.765625 0.453125q0.6875 0.4375 0.96875 1.234375q1.15625 -1.6875 2.984375 -1.6875q1.453125 0 2.21875 0.796875q0.78125 0.796875 0.78125 2.453125l0 6.640625l-1.640625 0l0 -6.09375q0 -0.984375 -0.15625 -1.40625q-0.15625 -0.4375 -0.578125 -0.703125q-0.421875 -0.265625 -0.984375 -0.265625q-1.015625 0 -1.6875 0.6875q-0.671875 0.671875 -0.671875 2.15625l0 5.625l-1.640625 0l0 -6.28125q0 -1.09375 -0.40625 -1.640625q-0.40625 -0.546875 -1.3125 -0.546875q-0.6875 0 -1.28125 0.359375q-0.59375 0.359375 -0.859375 1.0625q-0.25 0.703125 -0.25 2.03125l0 5.015625l-1.640625 0zm15.993927 0l0 -1.875l1.875 0l0 1.875l-1.875 0zm4.964554 0l0 -13.359375l9.656254 0l0 1.578125l-7.875004 0l0 4.09375l7.375004 0l0 1.5625l-7.375004 0l0 4.546875l8.187504 0l0 1.578125l-9.968754 0zm12.209202 0l0 -9.671875l1.46875 0l0 1.375q1.0625 -1.59375 3.078125 -1.59375q0.875 0 1.609375 0.3125q0.734375 0.3125 1.09375 0.828125q0.375 0.5 0.515625 1.203125q0.09375 0.453125 0.09375 1.59375l0 5.953125l-1.640625 0l0 -5.890625q0 -1.0 -0.203125 -1.484375q-0.1875 -0.5 -0.671875 -0.796875q-0.484375 -0.296875 -1.140625 -0.296875q-1.046875 0 -1.8125 0.671875q-0.75 0.65625 -0.75 2.515625l0 5.28125l-1.640625 0zm13.063217 0l-3.6875 -9.671875l1.734375 0l2.078125 5.796875q0.328125 0.9375 0.625 1.9375q0.203125 -0.765625 0.609375 -1.828125l2.140625 -5.90625l1.6875 0l-3.65625 9.671875l-1.53125 0z" fill-rule="nonzero"/><path fill="#ffffff" d="m166.6378 20.548557l58.204712 0l0 36.7874l-58.204712 0z" fill-rule="evenodd"/><path stroke="#000000" stroke-width="1.0" stroke-linejoin="round" stroke-linecap="butt" d="m166.6378 20.548557l58.204712 0l0 36.7874l-58.204712 0z" fill-rule="evenodd"/><path fill="#000000" d="m176.66905 43.742256l0 -9.546875l3.59375 0q0.953125 0 1.453125 0.09375q0.703125 0.125 1.171875 0.453125q0.484375 0.328125 0.765625 0.921875q0.296875 0.59375 0.296875 1.296875q0 1.21875 -0.78125 2.0625q-0.765625 0.84375 -2.796875 0.84375l-2.4375 0l0 3.875l-1.265625 0zm1.265625 -5.0l2.453125 0q1.234375 0 1.75 -0.453125q0.515625 -0.46875 0.515625 -1.28125q0 -0.609375 -0.3125 -1.03125q-0.296875 -0.421875 -0.796875 -0.5625q-0.3125 -0.09375 -1.171875 -0.09375l-2.4375 0l0 3.421875zm7.0303802 1.546875q0 -1.921875 1.078125 -2.84375q0.890625 -0.765625 2.171875 -0.765625q1.421875 0 2.328125 0.9375q0.90625 0.921875 0.90625 2.578125q0 1.328125 -0.40625 2.09375q-0.390625 0.765625 -1.15625 1.1875q-0.765625 0.421875 -1.671875 0.421875q-1.453125 0 -2.359375 -0.921875q-0.890625 -0.9375 -0.890625 -2.6875zm1.203125 0q0 1.328125 0.578125 1.984375q0.59375 0.65625 1.46875 0.65625q0.875 0 1.453125 -0.65625q0.578125 -0.671875 0.578125 -2.03125q0 -1.28125 -0.59375 -1.9375q-0.578125 -0.65625 -1.4375 -0.65625q-0.875 0 -1.46875 0.65625q-0.578125 0.65625 -0.578125 1.984375zm6.6312256 3.453125l0 -9.546875l1.171875 0l0 9.54Line truncated
|
||||
|
After Width: | Height: | Size: 35 KiB |
@@ -0,0 +1,138 @@
|
||||
RLlib Table of Contents
|
||||
=======================
|
||||
|
||||
Training APIs
|
||||
-------------
|
||||
* `Command-line <rllib-training.html>`__
|
||||
* `Configuration <rllib-training.html#configuration>`__
|
||||
* `Python API <rllib-training.html#python-api>`__
|
||||
* `Debugging <rllib-training.html#debugging>`__
|
||||
* `REST API <rllib-training.html#rest-api>`__
|
||||
|
||||
Environments
|
||||
------------
|
||||
* `RLlib Environments Overview <rllib-env.html>`__
|
||||
* `Feature Compatibility Matrix <rllib-env.html#feature-compatibility-matrix>`__
|
||||
* `OpenAI Gym <rllib-env.html#openai-gym>`__
|
||||
* `Vectorized <rllib-env.html#vectorized>`__
|
||||
* `Multi-Agent and Hierarchical <rllib-env.html#multi-agent-and-hierarchical>`__
|
||||
* `Interfacing with External Agents <rllib-env.html#interfacing-with-external-agents>`__
|
||||
* `Advanced Integrations <rllib-env.html#advanced-integrations>`__
|
||||
|
||||
Models, Preprocessors, and Action Distributions
|
||||
-----------------------------------------------
|
||||
* `RLlib Models, Preprocessors, and Action Distributions Overview <rllib-models.html>`__
|
||||
* `TensorFlow Models <rllib-models.html#tensorflow-models>`__
|
||||
* `PyTorch Models <rllib-models.html#pytorch-models>`__
|
||||
* `Custom Preprocessors <rllib-models.html#custom-preprocessors>`__
|
||||
* `Custom Action Distributions <rllib-models.html#custom-action-distributions>`__
|
||||
* `Supervised Model Losses <rllib-models.html#supervised-model-losses>`__
|
||||
* `Variable-length / Parametric Action Spaces <rllib-models.html#variable-length-parametric-action-spaces>`__
|
||||
* `Autoregressive Action Distributions <rllib-models.html#autoregressive-action-distributions>`__
|
||||
|
||||
Algorithms
|
||||
----------
|
||||
|
||||
* High-throughput architectures
|
||||
|
||||
- `Distributed Prioritized Experience Replay (Ape-X) <rllib-algorithms.html#distributed-prioritized-experience-replay-ape-x>`__
|
||||
|
||||
- `Importance Weighted Actor-Learner Architecture (IMPALA) <rllib-algorithms.html#importance-weighted-actor-learner-architecture-impala>`__
|
||||
|
||||
- `Asynchronous Proximal Policy Optimization (APPO) <rllib-algorithms.html#asynchronous-proximal-policy-optimization-appo>`__
|
||||
|
||||
* Gradient-based
|
||||
|
||||
- `Advantage Actor-Critic (A2C, A3C) <rllib-algorithms.html#advantage-actor-critic-a2c-a3c>`__
|
||||
|
||||
- `Deep Deterministic Policy Gradients (DDPG, TD3) <rllib-algorithms.html#deep-deterministic-policy-gradients-ddpg-td3>`__
|
||||
|
||||
- `Deep Q Networks (DQN, Rainbow, Parametric DQN) <rllib-algorithms.html#deep-q-networks-dqn-rainbow-parametric-dqn>`__
|
||||
|
||||
- `Policy Gradients <rllib-algorithms.html#policy-gradients>`__
|
||||
|
||||
- `Proximal Policy Optimization (PPO) <rllib-algorithms.html#proximal-policy-optimization-ppo>`__
|
||||
|
||||
- `Soft Actor Critic (SAC) <rllib-algorithms.html#soft-actor-critic-sac>`__
|
||||
|
||||
* Derivative-free
|
||||
|
||||
- `Augmented Random Search (ARS) <rllib-algorithms.html#augmented-random-search-ars>`__
|
||||
|
||||
- `Evolution Strategies <rllib-algorithms.html#evolution-strategies>`__
|
||||
|
||||
* Multi-agent specific
|
||||
|
||||
- `QMIX Monotonic Value Factorisation (QMIX, VDN, IQN) <rllib-algorithms.html#qmix-monotonic-value-factorisation-qmix-vdn-iqn>`__
|
||||
- `Multi-Agent Deep Deterministic Policy Gradient (contrib/MADDPG) <rllib-algorithms.html#multi-agent-deep-deterministic-policy-gradient-contrib-maddpg>`__
|
||||
|
||||
* Offline
|
||||
|
||||
- `Advantage Re-Weighted Imitation Learning (MARWIL) <rllib-algorithms.html#advantage-re-weighted-imitation-learning-marwil>`__
|
||||
|
||||
Offline Datasets
|
||||
----------------
|
||||
* `Working with Offline Datasets <rllib-offline.html>`__
|
||||
* `Input Pipeline for Supervised Losses <rllib-offline.html#input-pipeline-for-supervised-losses>`__
|
||||
* `Input API <rllib-offline.html#input-api>`__
|
||||
* `Output API <rllib-offline.html#output-api>`__
|
||||
|
||||
Concepts and Custom Algorithms
|
||||
------------------------------
|
||||
* `Policies <rllib-concepts.html>`__
|
||||
|
||||
- `Policies in Multi-Agent <rllib-concepts.html#policies-in-multi-agent>`__
|
||||
|
||||
- `Building Policies in TensorFlow <rllib-concepts.html#building-policies-in-tensorflow>`__
|
||||
|
||||
- `Building Policies in TensorFlow Eager <rllib-concepts.html#building-policies-in-tensorflow-eager>`__
|
||||
|
||||
- `Building Policies in PyTorch <rllib-concepts.html#building-policies-in-pytorch>`__
|
||||
|
||||
- `Extending Existing Policies <rllib-concepts.html#extending-existing-policies>`__
|
||||
|
||||
* `Policy Evaluation <rllib-concepts.html#policy-evaluation>`__
|
||||
* `Policy Optimization <rllib-concepts.html#policy-optimization>`__
|
||||
* `Trainers <rllib-concepts.html#trainers>`__
|
||||
|
||||
Examples
|
||||
--------
|
||||
|
||||
* `Tuned Examples <rllib-examples.html#tuned-examples>`__
|
||||
* `Training Workflows <rllib-examples.html#training-workflows>`__
|
||||
* `Custom Envs and Models <rllib-examples.html#custom-envs-and-models>`__
|
||||
* `Serving and Offline <rllib-examples.html#serving-and-offline>`__
|
||||
* `Multi-Agent and Hierarchical <rllib-examples.html#multi-agent-and-hierarchical>`__
|
||||
* `Community Examples <rllib-examples.html#community-examples>`__
|
||||
|
||||
Development
|
||||
-----------
|
||||
|
||||
* `Development Install <rllib-dev.html#development-install>`__
|
||||
* `API Stability <rllib-dev.html#api-stability>`__
|
||||
* `Features <rllib-dev.html#feature-development>`__
|
||||
* `Benchmarks <rllib-dev.html#benchmarks>`__
|
||||
* `Contributing Algorithms <rllib-dev.html#contributing-algorithms>`__
|
||||
|
||||
Package Reference
|
||||
-----------------
|
||||
* `ray.rllib.agents <rllib-package-ref.html#module-ray.rllib.agents>`__
|
||||
* `ray.rllib.env <rllib-package-ref.html#module-ray.rllib.env>`__
|
||||
* `ray.rllib.evaluation <rllib-package-ref.html#module-ray.rllib.evaluation>`__
|
||||
* `ray.rllib.models <rllib-package-ref.html#module-ray.rllib.models>`__
|
||||
* `ray.rllib.optimizers <rllib-package-ref.html#module-ray.rllib.optimizers>`__
|
||||
* `ray.rllib.utils <rllib-package-ref.html#module-ray.rllib.utils>`__
|
||||
|
||||
Troubleshooting
|
||||
---------------
|
||||
|
||||
If you encounter errors like
|
||||
`blas_thread_init: pthread_create: Resource temporarily unavailable` when using many workers,
|
||||
try setting ``OMP_NUM_THREADS=1``. Similarly, check configured system limits with
|
||||
`ulimit -a` for other resource limit errors.
|
||||
|
||||
If you encounter out-of-memory errors, consider setting ``redis_max_memory`` and ``object_store_memory`` in ``ray.init()`` to reduce memory usage.
|
||||
|
||||
For debugging unexpected hangs or performance problems, you can run ``ray stack`` to dump
|
||||
the stack traces of all Ray workers on the current node, and ``ray timeline`` to dump
|
||||
a timeline visualization of tasks to a file.
|
||||
+63
-109
@@ -7,155 +7,109 @@ RLlib is an open-source library for reinforcement learning that offers both high
|
||||
|
||||
To get started, take a look over the `custom env example <https://github.com/ray-project/ray/blob/master/rllib/examples/custom_env.py>`__ and the `API documentation <rllib-training.html>`__. If you're looking to develop custom algorithms with RLlib, also check out `concepts and custom algorithms <rllib-concepts.html>`__.
|
||||
|
||||
Installation
|
||||
------------
|
||||
RLlib in 60 seconds
|
||||
-------------------
|
||||
|
||||
The following is a whirlwind overview of RLlib. See also the full `table of contents <rllib-toc.html>`__ for a more in-depth guide including the `list of built-in algorithms <rllib-toc.html#algorithms>`__.
|
||||
|
||||
Running RLlib
|
||||
~~~~~~~~~~~~~
|
||||
|
||||
RLlib has extra dependencies on top of ``ray``. First, you'll need to install either `PyTorch <http://pytorch.org/>`__ or `TensorFlow <https://www.tensorflow.org>`__. Then, install the RLlib module:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
pip install tensorflow # or tensorflow-gpu
|
||||
pip install ray[rllib] # also recommended: ray[debug]
|
||||
|
||||
You might also want to clone the `Ray repo <https://github.com/ray-project/ray>`__ for convenient access to RLlib helper scripts:
|
||||
Then, you can try out training in the following equivalent ways:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
git clone https://github.com/ray-project/ray
|
||||
cd ray/rllib
|
||||
rllib train --run=PPO --env=CartPole-v0
|
||||
|
||||
Training APIs
|
||||
-------------
|
||||
* `Command-line <rllib-training.html>`__
|
||||
* `Configuration <rllib-training.html#configuration>`__
|
||||
* `Python API <rllib-training.html#python-api>`__
|
||||
* `Debugging <rllib-training.html#debugging>`__
|
||||
* `REST API <rllib-training.html#rest-api>`__
|
||||
.. code-block:: python
|
||||
|
||||
Environments
|
||||
------------
|
||||
* `RLlib Environments Overview <rllib-env.html>`__
|
||||
* `Feature Compatibility Matrix <rllib-env.html#feature-compatibility-matrix>`__
|
||||
* `OpenAI Gym <rllib-env.html#openai-gym>`__
|
||||
* `Vectorized <rllib-env.html#vectorized>`__
|
||||
* `Multi-Agent and Hierarchical <rllib-env.html#multi-agent-and-hierarchical>`__
|
||||
* `Interfacing with External Agents <rllib-env.html#interfacing-with-external-agents>`__
|
||||
* `Advanced Integrations <rllib-env.html#advanced-integrations>`__
|
||||
from ray import tune
|
||||
from ray.rllib.agents.ppo import PPOTrainer
|
||||
tune.run(PPOTrainer, config={"env": "CartPole-v0"})
|
||||
|
||||
Models, Preprocessors, and Action Distributions
|
||||
-----------------------------------------------
|
||||
* `RLlib Models, Preprocessors, and Action Distributions Overview <rllib-models.html>`__
|
||||
* `TensorFlow Models <rllib-models.html#tensorflow-models>`__
|
||||
* `PyTorch Models <rllib-models.html#pytorch-models>`__
|
||||
* `Custom Preprocessors <rllib-models.html#custom-preprocessors>`__
|
||||
* `Custom Action Distributions <rllib-models.html#custom-action-distributions>`__
|
||||
* `Supervised Model Losses <rllib-models.html#supervised-model-losses>`__
|
||||
* `Variable-length / Parametric Action Spaces <rllib-models.html#variable-length-parametric-action-spaces>`__
|
||||
* `Autoregressive Action Distributions <rllib-models.html#autoregressive-action-distributions>`__
|
||||
Next, we'll cover three key concepts in RLlib: Policies, Samples, and Trainers.
|
||||
|
||||
Algorithms
|
||||
----------
|
||||
Policies
|
||||
~~~~~~~~
|
||||
|
||||
* High-throughput architectures
|
||||
`Policies <rllib-concepts.html#policies>`__ are a core concept in RLlib. In a nutshell, policies are Python classes that define how an agent acts in an environment. `Rollout workers <rllib-concepts.html#policy-evaluation>`__ query the policy to determine agent actions. In a `gym <rllib-env.html#openai-gym>`__ environment, there is a single agent and policy. In `vector envs <rllib-env.html#vectorized>`__, policy inference is for multiple agents at once, and in `multi-agent <rllib-env.html#multi-agent-and-hierarchical>`__, there may be multiple policies, each controlling one or more agents:
|
||||
|
||||
- `Distributed Prioritized Experience Replay (Ape-X) <rllib-algorithms.html#distributed-prioritized-experience-replay-ape-x>`__
|
||||
.. image:: multi-flat.svg
|
||||
|
||||
- `Importance Weighted Actor-Learner Architecture (IMPALA) <rllib-algorithms.html#importance-weighted-actor-learner-architecture-impala>`__
|
||||
Policies can be implemented using `any framework <https://github.com/ray-project/ray/blob/master/rllib/policy/policy.py>`__. However, for TensorFlow and PyTorch, RLlib has `build_tf_policy <rllib-concepts.html#building-policies-in-tensorflow>`__ and `build_torch_policy <rllib-concepts.html#building-policies-in-pytorch>`__ helper functions that let you define a trainable policy with a functional-style API, for example:
|
||||
|
||||
- `Asynchronous Proximal Policy Optimization (APPO) <rllib-algorithms.html#asynchronous-proximal-policy-optimization-appo>`__
|
||||
.. code-block:: python
|
||||
|
||||
* Gradient-based
|
||||
def policy_gradient_loss(policy, batch_tensors):
|
||||
actions = batch_tensors[SampleBatch.ACTIONS]
|
||||
rewards = batch_tensors[SampleBatch.REWARDS]
|
||||
return -tf.reduce_mean(policy.action_dist.logp(actions) * rewards)
|
||||
|
||||
- `Advantage Actor-Critic (A2C, A3C) <rllib-algorithms.html#advantage-actor-critic-a2c-a3c>`__
|
||||
# <class 'ray.rllib.policy.tf_policy_template.MyTFPolicy'>
|
||||
MyTFPolicy = build_tf_policy(
|
||||
name="MyTFPolicy",
|
||||
loss_fn=policy_gradient_loss)
|
||||
|
||||
- `Deep Deterministic Policy Gradients (DDPG, TD3) <rllib-algorithms.html#deep-deterministic-policy-gradients-ddpg-td3>`__
|
||||
Sample Batches
|
||||
~~~~~~~~~~~~~~
|
||||
|
||||
- `Deep Q Networks (DQN, Rainbow, Parametric DQN) <rllib-algorithms.html#deep-q-networks-dqn-rainbow-parametric-dqn>`__
|
||||
Whether running in a single process or `large cluster <rllib-training.html#specifying-resources>`__, all data interchange in RLlib is in the form of `sample batches <https://github.com/ray-project/ray/blob/master/rllib/policy/sample_batch.py>`__. Sample batches encode one or more fragments of a trajectory. Typically, RLlib collects batches of size ``sample_batch_size`` from rollout workers, and concatenates one or more of these batches into a batch of size ``train_batch_size`` that is the input to SGD.
|
||||
|
||||
- `Policy Gradients <rllib-algorithms.html#policy-gradients>`__
|
||||
A typical sample batch looks something like the following when summarized. Since all values are kept in arrays, this allows for efficient encoding and transmission across the network:
|
||||
|
||||
- `Proximal Policy Optimization (PPO) <rllib-algorithms.html#proximal-policy-optimization-ppo>`__
|
||||
.. code-block:: python
|
||||
|
||||
- `Soft Actor Critic (SAC) <rllib-algorithms.html#soft-actor-critic-sac>`__
|
||||
{ 'action_logp': np.ndarray((200,), dtype=float32, min=-0.701, max=-0.685, mean=-0.694),
|
||||
'actions': np.ndarray((200,), dtype=int64, min=0.0, max=1.0, mean=0.495),
|
||||
'dones': np.ndarray((200,), dtype=bool, min=0.0, max=1.0, mean=0.055),
|
||||
'infos': np.ndarray((200,), dtype=object, head={}),
|
||||
'new_obs': np.ndarray((200, 4), dtype=float32, min=-2.46, max=2.259, mean=0.018),
|
||||
'obs': np.ndarray((200, 4), dtype=float32, min=-2.46, max=2.259, mean=0.016),
|
||||
'rewards': np.ndarray((200,), dtype=float32, min=1.0, max=1.0, mean=1.0),
|
||||
't': np.ndarray((200,), dtype=int64, min=0.0, max=34.0, mean=9.14)}
|
||||
|
||||
* Derivative-free
|
||||
In `multi-agent mode <rllib-concepts.html#policies-in-multi-agent>`__, sample batches are collected separately for each individual policy.
|
||||
|
||||
- `Augmented Random Search (ARS) <rllib-algorithms.html#augmented-random-search-ars>`__
|
||||
Training
|
||||
~~~~~~~~
|
||||
|
||||
- `Evolution Strategies <rllib-algorithms.html#evolution-strategies>`__
|
||||
Policies each define a ``learn_on_batch()`` method that improves the policy given a sample batch of input. For TF and Torch policies, this is implemented using a `loss function` that takes as input sample batch tensors and outputs a scalar loss. Here are a few example loss functions:
|
||||
|
||||
* Multi-agent specific
|
||||
- Simple `policy gradient loss <https://github.com/ray-project/ray/blob/master/rllib/agents/pg/pg_policy.py>`__
|
||||
- Simple `Q-function loss <https://github.com/ray-project/ray/blob/a1d2e1762325cd34e14dc411666d63bb15d6eaf0/rllib/agents/dqn/simple_q_policy.py#L136>`__
|
||||
- Importance-weighted `APPO surrogate loss <https://github.com/ray-project/ray/blob/master/rllib/agents/ppo/appo_policy.py>`__
|
||||
|
||||
- `QMIX Monotonic Value Factorisation (QMIX, VDN, IQN) <rllib-algorithms.html#qmix-monotonic-value-factorisation-qmix-vdn-iqn>`__
|
||||
- `Multi-Agent Deep Deterministic Policy Gradient (contrib/MADDPG) <rllib-algorithms.html#multi-agent-deep-deterministic-policy-gradient-contrib-maddpg>`__
|
||||
RLlib `Trainer classes <rllib-concepts.html#trainers>`__ coordinate the distributed workflow of running rollouts and optimizing policies. They do this by leveraging `policy optimizers <rllib-concepts.html#policy-optimization>`__ that implement the desired computation pattern (i.e., synchronous or asynchronous sampling, distributed replay, etc):
|
||||
|
||||
* Offline
|
||||
.. figure:: a2c-arch.svg
|
||||
|
||||
- `Advantage Re-Weighted Imitation Learning (MARWIL) <rllib-algorithms.html#advantage-re-weighted-imitation-learning-marwil>`__
|
||||
Synchronous Sampling (e.g., A2C, PG, PPO)
|
||||
|
||||
Offline Datasets
|
||||
----------------
|
||||
* `Working with Offline Datasets <rllib-offline.html>`__
|
||||
* `Input Pipeline for Supervised Losses <rllib-offline.html#input-pipeline-for-supervised-losses>`__
|
||||
* `Input API <rllib-offline.html#input-api>`__
|
||||
* `Output API <rllib-offline.html#output-api>`__
|
||||
.. figure:: dqn-arch.svg
|
||||
|
||||
Concepts and Custom Algorithms
|
||||
------------------------------
|
||||
* `Policies <rllib-concepts.html>`__
|
||||
Synchronous Replay (e.g., DQN, DDPG, TD3)
|
||||
|
||||
- `Policies in Multi-Agent <rllib-concepts.html#policies-in-multi-agent>`__
|
||||
.. figure:: impala-arch.svg
|
||||
|
||||
- `Building Policies in TensorFlow <rllib-concepts.html#building-policies-in-tensorflow>`__
|
||||
Asynchronous Sampling (e.g., IMPALA, APPO)
|
||||
|
||||
- `Building Policies in TensorFlow Eager <rllib-concepts.html#building-policies-in-tensorflow-eager>`__
|
||||
.. figure:: apex-arch.svg
|
||||
|
||||
- `Building Policies in PyTorch <rllib-concepts.html#building-policies-in-pytorch>`__
|
||||
Asynchronous Replay (e.g., Ape-X)
|
||||
|
||||
- `Extending Existing Policies <rllib-concepts.html#extending-existing-policies>`__
|
||||
RLlib uses `Ray actors <actors.html>`__ to scale these architectures from a single core to many thousands of cores in a cluster. You can `configure the parallelism <rllib-training.html#specifying-resources>`__ used for training by changing the ``num_workers`` parameter.
|
||||
|
||||
* `Policy Evaluation <rllib-concepts.html#policy-evaluation>`__
|
||||
* `Policy Optimization <rllib-concepts.html#policy-optimization>`__
|
||||
* `Trainers <rllib-concepts.html#trainers>`__
|
||||
Customization
|
||||
~~~~~~~~~~~~~
|
||||
|
||||
Examples
|
||||
--------
|
||||
RLlib provides ways to customize almost all aspects of training, including the `environment <rllib-env.html#configuring-environments>`__, `neural network model <rllib-models.html#tensorflow-models>`__, `action distribution <rllib-models.html#custom-action-distributions>`__, and `policy definitions <rllib-concepts.html#policies>`__:
|
||||
|
||||
* `Tuned Examples <rllib-examples.html#tuned-examples>`__
|
||||
* `Training Workflows <rllib-examples.html#training-workflows>`__
|
||||
* `Custom Envs and Models <rllib-examples.html#custom-envs-and-models>`__
|
||||
* `Serving and Offline <rllib-examples.html#serving-and-offline>`__
|
||||
* `Multi-Agent and Hierarchical <rllib-examples.html#multi-agent-and-hierarchical>`__
|
||||
* `Community Examples <rllib-examples.html#community-examples>`__
|
||||
.. image:: rllib-components.svg
|
||||
|
||||
Development
|
||||
-----------
|
||||
|
||||
* `Development Install <rllib-dev.html#development-install>`__
|
||||
* `API Stability <rllib-dev.html#api-stability>`__
|
||||
* `Features <rllib-dev.html#feature-development>`__
|
||||
* `Benchmarks <rllib-dev.html#benchmarks>`__
|
||||
* `Contributing Algorithms <rllib-dev.html#contributing-algorithms>`__
|
||||
|
||||
Package Reference
|
||||
-----------------
|
||||
* `ray.rllib.agents <rllib-package-ref.html#module-ray.rllib.agents>`__
|
||||
* `ray.rllib.env <rllib-package-ref.html#module-ray.rllib.env>`__
|
||||
* `ray.rllib.evaluation <rllib-package-ref.html#module-ray.rllib.evaluation>`__
|
||||
* `ray.rllib.models <rllib-package-ref.html#module-ray.rllib.models>`__
|
||||
* `ray.rllib.optimizers <rllib-package-ref.html#module-ray.rllib.optimizers>`__
|
||||
* `ray.rllib.utils <rllib-package-ref.html#module-ray.rllib.utils>`__
|
||||
|
||||
Troubleshooting
|
||||
---------------
|
||||
|
||||
If you encounter errors like
|
||||
`blas_thread_init: pthread_create: Resource temporarily unavailable` when using many workers,
|
||||
try setting ``OMP_NUM_THREADS=1``. Similarly, check configured system limits with
|
||||
`ulimit -a` for other resource limit errors.
|
||||
|
||||
If you encounter out-of-memory errors, consider setting ``redis_max_memory`` and ``object_store_memory`` in ``ray.init()`` to reduce memory usage.
|
||||
|
||||
For debugging unexpected hangs or performance problems, you can run ``ray stack`` to dump
|
||||
the stack traces of all Ray workers on the current node, and ``ray timeline`` to dump
|
||||
a timeline visualization of tasks to a file.
|
||||
To learn more, proceed to the `table of contents <rllib-toc.html>`__.
|
||||
Reference in new issue
Block a user