mirror of
https://github.com/wassname/ray.git
synced 2026-08-09 12:20:09 +08:00
[RLlib] Issue 9071 A3C w/ RNN not working due to VF assuming no RNN. (#13238)
This commit is contained in:
@@ -1,22 +1,12 @@
|
||||
import numpy as np
|
||||
import scipy.signal
|
||||
from typing import Dict, Optional
|
||||
|
||||
from ray.rllib.evaluation.episode import MultiAgentEpisode
|
||||
from ray.rllib.policy.policy import Policy
|
||||
from ray.rllib.policy.sample_batch import SampleBatch
|
||||
from ray.rllib.utils.annotations import DeveloperAPI
|
||||
|
||||
|
||||
def discount_cumsum(x: np.ndarray, gamma: float) -> float:
|
||||
"""Calculates the discounted cumulative sum over a reward sequence `x`.
|
||||
|
||||
y[t] - discount*y[t+1] = x[t]
|
||||
reversed(y)[t] - discount*reversed(y)[t-1] = reversed(x)[t]
|
||||
|
||||
Args:
|
||||
gamma (float): The discount factor gamma.
|
||||
|
||||
Returns:
|
||||
float: The discounted cumulative sum over the reward sequence `x`.
|
||||
"""
|
||||
return scipy.signal.lfilter([1], [1, float(-gamma)], x[::-1], axis=0)[::-1]
|
||||
from ray.rllib.utils.typing import AgentID
|
||||
|
||||
|
||||
class Postprocessing:
|
||||
@@ -89,3 +79,83 @@ def compute_advantages(rollout: SampleBatch,
|
||||
Postprocessing.ADVANTAGES].astype(np.float32)
|
||||
|
||||
return rollout
|
||||
|
||||
|
||||
def compute_gae_for_sample_batch(
|
||||
policy: Policy,
|
||||
sample_batch: SampleBatch,
|
||||
other_agent_batches: Optional[Dict[AgentID, SampleBatch]] = None,
|
||||
episode: Optional[MultiAgentEpisode] = None) -> SampleBatch:
|
||||
"""Adds GAE (generalized advantage estimations) to a trajectory.
|
||||
|
||||
The trajectory contains only data from one episode and from one agent.
|
||||
- If `config.batch_mode=truncate_episodes` (default), sample_batch may
|
||||
contain a truncated (at-the-end) episode, in case the
|
||||
`config.rollout_fragment_length` was reached by the sampler.
|
||||
- If `config.batch_mode=complete_episodes`, sample_batch will contain
|
||||
exactly one episode (no matter how long).
|
||||
New columns can be added to sample_batch and existing ones may be altered.
|
||||
|
||||
Args:
|
||||
policy (Policy): The Policy used to generate the trajectory
|
||||
(`sample_batch`)
|
||||
sample_batch (SampleBatch): The SampleBatch to postprocess.
|
||||
other_agent_batches (Optional[Dict[PolicyID, SampleBatch]]): Optional
|
||||
dict of AgentIDs mapping to other agents' trajectory data (from the
|
||||
same episode). NOTE: The other agents use the same policy.
|
||||
episode (Optional[MultiAgentEpisode]): Optional multi-agent episode
|
||||
object in which the agents operated.
|
||||
|
||||
Returns:
|
||||
SampleBatch: The postprocessed, modified SampleBatch (or a new one).
|
||||
"""
|
||||
|
||||
# Trajectory is actually complete -> last r=0.0.
|
||||
if sample_batch[SampleBatch.DONES][-1]:
|
||||
last_r = 0.0
|
||||
# Trajectory has been truncated -> last r=VF estimate of last obs.
|
||||
else:
|
||||
# Input dict is provided to us automatically via the Model's
|
||||
# requirements. It's a single-timestep (last one in trajectory)
|
||||
# input_dict.
|
||||
if policy.config.get("_use_trajectory_view_api"):
|
||||
# Create an input dict according to the Model's requirements.
|
||||
input_dict = policy.model.get_input_dict(
|
||||
sample_batch, index="last")
|
||||
last_r = policy._value(**input_dict)
|
||||
# TODO: (sven) Remove once trajectory view API is all-algo default.
|
||||
else:
|
||||
next_state = []
|
||||
for i in range(policy.num_state_tensors()):
|
||||
next_state.append(sample_batch["state_out_{}".format(i)][-1])
|
||||
last_r = policy._value(sample_batch[SampleBatch.NEXT_OBS][-1],
|
||||
sample_batch[SampleBatch.ACTIONS][-1],
|
||||
sample_batch[SampleBatch.REWARDS][-1],
|
||||
*next_state)
|
||||
|
||||
# Adds the policy logits, VF preds, and advantages to the batch,
|
||||
# using GAE ("generalized advantage estimation") or not.
|
||||
batch = compute_advantages(
|
||||
sample_batch,
|
||||
last_r,
|
||||
policy.config["gamma"],
|
||||
policy.config["lambda"],
|
||||
use_gae=policy.config["use_gae"],
|
||||
use_critic=policy.config.get("use_critic", True))
|
||||
|
||||
return batch
|
||||
|
||||
|
||||
def discount_cumsum(x: np.ndarray, gamma: float) -> float:
|
||||
"""Calculates the discounted cumulative sum over a reward sequence `x`.
|
||||
|
||||
y[t] - discount*y[t+1] = x[t]
|
||||
reversed(y)[t] - discount*reversed(y)[t-1] = reversed(x)[t]
|
||||
|
||||
Args:
|
||||
gamma (float): The discount factor gamma.
|
||||
|
||||
Returns:
|
||||
float: The discounted cumulative sum over the reward sequence `x`.
|
||||
"""
|
||||
return scipy.signal.lfilter([1], [1, float(-gamma)], x[::-1], axis=0)[::-1]
|
||||
|
||||
Reference in New Issue
Block a user