From 3023d07290c5b17e7bd3032b53fe8f57be5b678d Mon Sep 17 00:00:00 2001 From: Brian Delhaisse Date: Fri, 12 Apr 2019 02:50:26 +0200 Subject: [PATCH] update storages: refactor ER as DictStorage --- pyrobolearn/storages/er.py | 370 +++++++++++++++++++++++++++----- pyrobolearn/storages/storage.py | 28 +-- 2 files changed, 330 insertions(+), 68 deletions(-) diff --git a/pyrobolearn/storages/er.py b/pyrobolearn/storages/er.py index c09d965..b570a17 100644 --- a/pyrobolearn/storages/er.py +++ b/pyrobolearn/storages/er.py @@ -7,9 +7,10 @@ References: """ import random +import numpy as np import torch -from pyrobolearn.storages.storage import Storage, ListStorage +from pyrobolearn.storages.storage import Storage, ListStorage, Batch, DictStorage __author__ = "Brian Delhaisse" @@ -22,8 +23,143 @@ __email__ = "briandelhaisse@gmail.com" __status__ = "Development" -class ExperienceReplay(Storage): - r"""Experience replay storage +# class ExperienceReplay(Storage): +# r"""Experience replay storage +# +# The experience replay storage returns a transition tuple :math:`(s_t, a_t, s_{t+1}, r_t, d)`, where is :math:`s_t` +# is the state at time :math:`t`, :math:`a_t` is the action outputted by the policy in response to the state +# :math:`s_t`, :math:`s_{t+1}` is the next state returned by the environment due to the policy's action :math:`a_t` +# and the current state :math:`s_t`, :math:`r_t` is the reward signal returned by the environment, and :math:`d` +# is a boolean value that specifies if the task is over or not (i.e. if it has failed or succeeded). +# +# The experience replay storage is often used in conjunction with off-policy RL algorithms. +# +# The following code is inspired by [3] but modified such that it uses a PyTorch list storage. +# +# References: +# [1] "Reinforcement Learning for robots using neural networks", Lin, 1993 +# [2] "Playing Atari with Deep Reinforcement Learning", Mnih et al., 2013 +# """ +# +# def __init__(self, capacity=10000, device=None, dtype=torch.float): +# """ +# Initialize the Experience Replay Storage. +# +# Args: +# capacity (int): maximum size of the experience replay storage. +# device (torch.device, str, None): the device to put the data on (e.g. `torch.device("cuda:0")` or +# `torch.device("cpu")`). If string, it can be 'cpu' or 'cuda'. If None, it will keep the original device +# to which the tensor is allocated. +# dtype (torch.dtype, None): convert the `torch.Tensor` to the specified data type. If None, it will keep +# the original dtype +# """ +# super(ExperienceReplay, self).__init__(device, dtype) +# if capacity <= 0: +# raise ValueError("Expecting the capacity to be bigger than 0, instead got: {}".format(capacity)) +# self.capacity = capacity +# self.memory = ListStorage(device=device, dtype=dtype) +# self.position = 0 +# +# ############## +# # Properties # +# ############## +# +# @property +# def device(self): +# """Return the memory's device instance.""" +# return self.memory.device +# +# @device.setter +# def device(self, device): +# """Set the memory's device instance.""" +# self.memory.device = device +# +# @property +# def dtype(self): +# """Return the memory's data type.""" +# return self.memory.dtype +# +# @dtype.setter +# def dtype(self, dtype): +# """Set the memory's data type.""" +# self.memory.dtype = dtype +# +# @property +# def size(self): +# """Return the size of the experience replay storage, i.e. the number of transition tuples.""" +# return len(self.memory) +# +# ########### +# # Methods # +# ########### +# +# def push(self, *items): +# r""" +# Push new transition :math:`(s_t, a_t, s_{t+1}, r_t, d)` in the experience replay. +# +# Args: +# *items (list, tuple of torch.Tensor): transition tuple +# """ +# # add new item or update previous one +# if len(self.memory) < self.capacity: +# self.memory.append(items) +# else: +# self.memory[self.position] = items +# +# # update head position (cyclic) +# self.position = (self.position + 1) % self.capacity +# +# def to(self, device=None, dtype=None): +# """ +# Put all the tensors to the specified device and convert them to the specified data type. +# +# Args: +# device (torch.device, str, None): the device to put the data on (e.g. `torch.device("cuda:0")` or +# `torch.device("cpu")`). If string, it can be 'cpu' or 'cuda'. If None, it will keep the original device +# to which the tensor is allocated. +# dtype (torch.dtype, None): convert the `torch.Tensor` to the specified data type. If None, it will keep +# the original dtype +# """ +# self.memory.to(device=device, dtype=dtype) +# +# def sample(self, batch_size): +# """Sample uniformly a batch from the experience replay.""" +# return random.sample(self, batch_size) +# +# def get_batch(self, indices): +# """Return a batch of the transitions in the form of a `DictStorage`. +# +# Args: +# indices (list of int): indices. Each index must be between 0 and `self.size`. +# +# Returns: +# DictStorage / Batch: batch containing a part of the storage. Variables such as `observations`, `actions`, +# `rewards`, `next_states`, `masks`, and others can be accessed from the object. +# """ +# # create batch +# batch = {} +# +# # check indices +# indices = np.array(indices) +# indices = indices[indices < self.size] +# +# for idx in indices: +# transition = self.memory[idx] +# batch.setdefault('observations', []).append(transition[0]) +# batch.setdefault('actions', []).append(transition[1]) +# batch.setdefault('next_states', []).append(transition[2]) +# batch.setdefault('rewards', []).append(transition[3]) +# batch.setdefault('masks', []).append(transition[4]) +# +# return Batch(batch, device=self.device, dtype=self.dtype) +# +# +# # alias +# ER = ExperienceReplay + + +class ExperienceReplay(DictStorage): # ExperienceReplayStorage(DictStorage): + r"""Experience Replay Storage The experience replay storage returns a transition tuple :math:`(s_t, a_t, s_{t+1}, r_t, d)`, where is :math:`s_t` is the state at time :math:`t`, :math:`a_t` is the action outputted by the policy in response to the state @@ -40,84 +176,210 @@ class ExperienceReplay(Storage): [2] "Playing Atari with Deep Reinforcement Learning", Mnih et al., 2013 """ - def __init__(self, capacity=10000, device=None, dtype=torch.float): + def __init__(self, observation_shapes, action_shapes, capacity=10000, *args, **kwargs): """ - Initialize the Experience Replay Storage. + Initialize the experience replay storage. Args: + observation_shapes (list of tuple of int, tuple of int): each tuple represents the shape of an observation. + action_shapes (list of tuple of int, tuple of int): each tuple represents the shape of an action. capacity (int): maximum size of the experience replay storage. - device (torch.device, str, None): the device to put the data on (e.g. `torch.device("cuda:0")` or - `torch.device("cpu")`). If string, it can be 'cpu' or 'cuda'. If None, it will keep the original device - to which the tensor is allocated. - dtype (torch.dtype, None): convert the `torch.Tensor` to the specified data type. If None, it will keep - the original dtype """ - super(ExperienceReplay, self).__init__(device, dtype) - self.capacity = capacity - self.memory = ListStorage(device=device, dtype=dtype) + # recurrent_hidden_state_size (int): size of the internal state + print("\nStorage: observation shape: {}".format(observation_shapes)) + print("Storage: action shape: {}".format(action_shapes)) + super(ExperienceReplay, self).__init__() self.position = 0 + if capacity <= 0: + raise ValueError("Expecting the capacity to be bigger than 0, instead got: {}".format(capacity)) + self.capacity = capacity + self.full = False + self.init(observation_shapes, action_shapes) ############## # Properties # ############## @property - def device(self): - """Return the memory's device instance.""" - return self.memory.device - - @device.setter - def device(self, device): - """Set the memory's device instance.""" - self.memory.device = device - - @property - def dtype(self): - """Return the memory's data type.""" - return self.memory.dtype - - @dtype.setter - def dtype(self, dtype): - """Set the memory's data type.""" - self.memory.dtype = dtype + def size(self): + """Return the size of the experience replay storage.""" + if self.full: + return self.capacity + return self.position ########### # Methods # ########### - def push(self, *items): - r""" - Push new transition :math:`(s_t, a_t, s_{t+1}, r_t, d, \gamma)` in the experience replay. + def create_new_entry(self, key, shapes, dtype=torch.dtype): + """Create a new entry (=tensor) in the experience replay storage dictionary. The tensor will have the dimension + (num_steps, self.num_processes, *shape) for each shape in shapes, and will be initialized to zero. + The tensor will also have the same type than the other tensors and will be sent to the correct device. Args: - *items (list, tuple of torch.Tensor): transition tuple + key (str, object): key of the dictionary. + shapes (list of tuple of int, tuple of int, int): (list of) shape(s) of the tensor(s). + dtype (torch.dtype, np.generic, object): specify if we want to allocate a `torch.Tensor`, or a numpy array. + If dtype == torch.dtype, then it will be set to dtype = self.dtype. """ - # add new item or update previous one - if len(self.memory) < self.capacity: - self.memory.append(items) + # allocate the new tensor + + # convert torch type if necessary + if dtype == torch.dtype: + dtype = self.dtype + + # if we have a list of shapes + if isinstance(shapes, list): + if isinstance(dtype, torch.dtype): + self[key] = [torch.zeros(self.capacity, *shape).to(device=self.device, dtype=dtype) + for shape in shapes] + else: # numpy array + self[key] = [np.zeros((self.capacity,) + shape, dtype=dtype) for shape in shapes] + + # if the 'shapes' is a tuple + elif isinstance(shapes, tuple): + if isinstance(dtype, torch.dtype): + self[key] = torch.zeros(self.capacity, *shapes).to(device=self.device, dtype=dtype) + else: + self[key] = np.zeros((self.capacity,) + shapes, dtype=dtype) + + # if the 'shapes' is an int + elif isinstance(shapes, int): + if isinstance(dtype, torch.dtype): + self[key] = torch.zeros(self.capacity, shapes).to(device=self.device, dtype=dtype) + else: + self[key] = np.zeros((self.capacity, shapes), dtype=dtype) + else: - self.memory[self.position] = items + raise TypeError("Expecting the given shapes {} to be a list of tuple of int, a tuple of int, or an int, " + "instead got: {}".format({}, type(shapes))) + + def init(self, observation_shapes, action_shapes): + """ + Initialize the experience replay storage by allocating the appropriate tensors for the observations (states), + actions, next states, rewards, and masks. + + Args: + observation_shapes (list of tuple of int, tuple of int): each tuple represents the shape of an observation. + action_shapes (list of tuple of int, tuple of int): each tuple represents the shape of an action. + """ + # clear itself: remove all items from the DictStorage, and reset all variables + self.reset() + + # allocate space for observations / states + if not isinstance(observation_shapes, list): + observation_shapes = [observation_shapes] + self.create_new_entry('observations', shapes=observation_shapes) + + # allocate space for actions + if not isinstance(action_shapes, list): + action_shapes = [action_shapes] + self.create_new_entry('actions', shapes=action_shapes) + + # allocate space for next observations / states # TODO: improve this, don't copy, just keep reference + if not isinstance(observation_shapes, list): + observation_shapes = [observation_shapes] + self.create_new_entry('next_states', shapes=observation_shapes) + + # allocate space for rewards + self.create_new_entry('rewards', shapes=1) + + # allocate space for the masks + self.create_new_entry('masks', shapes=1) + + # space for log probabilities on policy, distributions, scalar values from value functions, + # recurrent hidden states, and others have to be allocated outside the class + + def reset(self): + """Reset the experience replay storage.""" + pass + + def clear(self): + """Clear the experience replay storage.""" + self.full = False + super(ExperienceReplayStorage, self).clear() + + def insert(self, observations, actions, reward, mask, next_states, **kwargs): + """ + Insert the given parameters into the storage. + + Args: + observations (torch.Tensor, list of torch.Tensor): (list of) state(s) / observation(s). + actions (torch.Tensor, list of torch.Tensor): (list of) action(s). + reward (float, int, torch.Tensor): reward value. + mask (float, int, torch.Tensor): masks. They are set to zeros after an episode has terminated. + next_states (torch.Tensor, list of torch.Tensor): (list of) next state(s) / observation(s). + **kwargs (dict): kwargs + """ + print("ER - insert observation: {}".format(observations)) + print("ER - insert action: {}".format(actions)) + print("ER - insert next state: {}".format(next_states)) + print("ER - insert reward: {}".format(reward)) + print("ER - insert mask: {}".format(mask)) + + # check given observations and actions + if not isinstance(observations, list): + observations = [observations] + if not isinstance(actions, list): + actions = [actions] + if not isinstance(next_states, list): + next_states = [next_states] + + # insert each observation / action / next states + for observation, storage in zip(observations, self.observations): + storage[self.position].copy_(self._convert_to_tensor(observation)) + for action, storage in zip(actions, self.actions): + storage[self.position].copy_(self._convert_to_tensor(action)) + for next_state, storage in zip(next_states, self.next_states): + storage[self.position].copy_(self._convert_to_tensor(next_state)) + + # insert reward and mask + self.rewards[self.position].copy_(self._convert_to_tensor(reward)) + self.masks[self.position].copy_(self._convert_to_tensor(mask)) # update head position (cyclic) - self.position = (self.position + 1) % self.capacity + self.position += 1 + if self.position >= self.capacity: + self.full = True + self.position = self.position % self.capacity - def to(self, device=None, dtype=None): - """ - Put all the tensors to the specified device and convert them to the specified data type. + def get_batch(self, indices): + """Return a batch of the experience replay storage in the form of a `DictStorage`. Args: - device (torch.device, str, None): the device to put the data on (e.g. `torch.device("cuda:0")` or - `torch.device("cpu")`). If string, it can be 'cpu' or 'cuda'. If None, it will keep the original device - to which the tensor is allocated. - dtype (torch.dtype, None): convert the `torch.Tensor` to the specified data type. If None, it will keep - the original dtype + indices (list of int): indices. Each index must be between 0 and `self.size`. + + Returns: + DictStorage / Batch: batch containing a part of the storage. Variables such as `observations`, `actions`, + `next_states`, `rewards`, and `masks`. """ - self.memory.to(device=device, dtype=dtype) + batch = {} - def sample(self, batch_size): - """Sample uniformly a batch from the experience replay.""" - return random.sample(self, batch_size) + # go through each attribute in the and sample from the tensors + for key, value in self.iteritems(): + if isinstance(list, value): # value = list of tensors + batch[key] = [val[indices] for val in value] + else: # value = tensor + batch[key] = value[indices] + # return batch (which is given to the updater (and loss)) + return Batch(batch, device=self.device, dtype=self.dtype) -# alias -ER = ExperienceReplay + ############# + # Operators # + ############# + + def __setitem__(self, key, value): + """Add the new value in the dictionary. Key must be strings or objects.""" + # if not isinstance(key, str): + # raise TypeError("The experience replay storage only accepts key as strings! Instead got: {} with type " + # "{}".format(key, type(key))) + super(ExperienceReplayStorage, self).__setitem__(key, value) + + # def __setattr__(self, key, value): + # """Set the attribute using the given key and value. That is, instead of `D[key] = value`, you can do + # `D.key = value`. By default, this creates a tensor with shape (num_steps + 1, self.num_processes, 1). + # + # Warnings: avoid to use this. + # """ + # self.create_new_entry(key, shapes=1) diff --git a/pyrobolearn/storages/storage.py b/pyrobolearn/storages/storage.py index 9ce18c6..60ac78e 100644 --- a/pyrobolearn/storages/storage.py +++ b/pyrobolearn/storages/storage.py @@ -641,7 +641,7 @@ class RolloutStorage(DictStorage): The tensor will also have the same type than the other tensors and will be sent to the correct device. Args: - key (str): key of the dictionary. + key (str, object): key of the dictionary. shapes (list of tuple of int, tuple of int, int): (list of) shape(s) of the tensor(s). num_steps (int, None): the number of time steps. This value can not be smaller than `self.num_steps`. If None, `self.num_steps` will be used. @@ -649,8 +649,8 @@ class RolloutStorage(DictStorage): If dtype == torch.dtype, then it will be set to dtype = self.dtype. """ # check the key - if not isinstance(key, str): - raise TypeError("Expecting the key to be a string, instead got: {} with type {}".format(key, type(key))) + # if not isinstance(key, str): + # raise TypeError("Expecting the key to be a string, instead got: {} with type {}".format(key, type(key))) # check the number of time steps if num_steps is None: @@ -814,7 +814,7 @@ class RolloutStorage(DictStorage): else: set_tensor(self[key], step, values, copy=copy) - def insert(self, observations, actions, rewards, masks, distributions=None, update_step=True, **kwargs): + def insert(self, observations, actions, reward, mask, distributions=None, update_step=True, **kwargs): # distributions, values=None): # recurrent_hidden_state, action_log_prob): """ @@ -823,8 +823,8 @@ class RolloutStorage(DictStorage): Args: observations (torch.Tensor, list of torch.Tensor): (list of) state(s) / observation(s). actions (torch.Tensor, list of torch.Tensor): (list of) action(s) - rewards (float, int, torch.Tensor): reward value - masks (float, int, torch.Tensor): masks. They are set to zeros after an episode has terminated. + reward (float, int, torch.Tensor): reward value + mask (float, int, torch.Tensor): masks. They are set to zeros after an episode has terminated. distributions (torch.distributions.Distribution, None): action distribution. update_step (bool): if True, it will update the current time step. If False, the user needs to call `step()` in order to update it. @@ -849,10 +849,10 @@ class RolloutStorage(DictStorage): storage[self._step].copy_(self._convert_to_tensor(action)) # insert rewards and masks - self.rewards[self._step].copy_(self._convert_to_tensor(rewards)) - if masks is None: - masks = torch.tensor(1.) - self.masks[self._step + 1].copy_(self._convert_to_tensor(masks)) + self.rewards[self._step].copy_(self._convert_to_tensor(reward)) + if mask is None: + mask = torch.tensor(1.) + self.masks[self._step + 1].copy_(self._convert_to_tensor(mask)) # insert distributions for distribution, storage in zip(distributions, self.distributions): @@ -908,10 +908,10 @@ class RolloutStorage(DictStorage): ############# def __setitem__(self, key, value): - """Add the new value in the dictionary. Key must be strings.""" - if not isinstance(key, str): - raise TypeError("The rollout storage only accepts key as strings! Instead got: {} with type " - "{}".format(key, type(key))) + """Add the new value in the dictionary. Key must be strings or objects.""" + # if not isinstance(key, str): + # raise TypeError("The rollout storage only accepts key as strings! Instead got: {} with type " + # "{}".format(key, type(key))) super(RolloutStorage, self).__setitem__(key, value) # def __setattr__(self, key, value):