From 61471fdf9c1654bc3b2130c43956e03170982bd5 Mon Sep 17 00:00:00 2001 From: Whi Kwon <30768596+whikwon@users.noreply.github.com> Date: Fri, 19 Jul 2019 00:11:29 +0900 Subject: [PATCH 1/3] Fix dockerfile (#84) * Fix dockerfile * Fix wrong filename --- Dockerfile | 4 ++-- launch/open_manipulator_env.launch | 6 +++--- 2 files changed, 5 insertions(+), 5 deletions(-) diff --git a/Dockerfile b/Dockerfile index a07cf05..05dc3c1 100644 --- a/Dockerfile +++ b/Dockerfile @@ -8,7 +8,7 @@ RUN apt-get update && apt-get -y upgrade && apt-get -y install git wget vim # install ROS RUN sh -c 'echo "deb http://packages.ros.org/ros/ubuntu xenial main" > /etc/apt/sources.list.d/ros-latest.list' -RUN sudo -E apt-key adv --keyserver 'hkp://keyserver.ubuntu.com:80' --recv-key C1CF6E31E6BADE8868B172B4F42ED6FBAB17C654 +RUN apt-key adv --keyserver 'hkp://keyserver.ubuntu.com:80' --recv-key C1CF6E31E6BADE8868B172B4F42ED6FBAB17C654 RUN apt-get update -y && apt-get upgrade -y RUN apt-get install -y ros-${ROS_DISTRO}-desktop-full ros-${ROS_DISTRO}-rqt-* RUN rosdep init && rosdep update @@ -49,7 +49,7 @@ RUN apt-get update && apt-get install -y python3-opengl zlib1g-dev libjpeg-dev p cmake swig libboost-all-dev libsdl2-dev libosmesa6-dev xvfb ffmpeg # install repository requirements -RUN apt-get remove -y python-psutil +RUN dpkg -P --force-all python-psutil python-enum34 python-yaml RUN cd src/kair_algorithms_draft/scripts && python2.7 -m pip install -r requirements.txt RUN python2.7 -m pip install gym['Box2d'] diff --git a/launch/open_manipulator_env.launch b/launch/open_manipulator_env.launch index 4ed66c1..8fd002a 100755 --- a/launch/open_manipulator_env.launch +++ b/launch/open_manipulator_env.launch @@ -8,9 +8,9 @@ - + - + @@ -25,7 +25,7 @@ - + Date: Sat, 20 Jul 2019 17:36:01 +0900 Subject: [PATCH 2/3] Add HER on sac (#85) * Add HER on sac * Add args in wandb log --- .gitignore | 5 +- scripts/algorithms/common/abstract/her.py | 83 +++++++++++++++++++ .../algorithms/common/abstract/reward_fn.py | 19 +++++ scripts/algorithms/sac/agent.py | 60 +++++++++++--- scripts/algorithms/td3/agent.py | 1 + .../agent/lunarlander_continuous_v2/sac.py | 13 ++- .../agent/lunarlander_continuous_v2/utils.py | 67 +++++++++++++++ .../config/environment/open_manipulator.py | 1 - .../open_manipulator_reacher_env.py | 4 +- .../envs/open_manipulator/ros_interface.py | 4 +- scripts/gazebo_test_open_manipulator.py | 2 +- scripts/run_open_manipulator_reacher_v0.py | 3 +- 12 files changed, 243 insertions(+), 19 deletions(-) create mode 100644 scripts/algorithms/common/abstract/her.py create mode 100644 scripts/algorithms/common/abstract/reward_fn.py create mode 100644 scripts/config/agent/lunarlander_continuous_v2/utils.py diff --git a/.gitignore b/.gitignore index df6e790..8d128d7 100644 --- a/.gitignore +++ b/.gitignore @@ -254,4 +254,7 @@ wandb save # pycharm -.idea \ No newline at end of file +.idea + +# vscode +.vscode diff --git a/scripts/algorithms/common/abstract/her.py b/scripts/algorithms/common/abstract/her.py new file mode 100644 index 0000000..9678ba9 --- /dev/null +++ b/scripts/algorithms/common/abstract/her.py @@ -0,0 +1,83 @@ +# -*- coding: utf-8 -*- +"""Abstract class used for Hindsight Experience Replay. +- Author: Kh Kim +- Contact: kh.kim@medipixel.io +- Paper: https://arxiv.org/pdf/1707.01495.pdf +""" + +from abc import ABCMeta, abstractmethod + +import numpy as np + + +class HER(object): + """Abstract class for HER (final strategy). + Attributes: + reward_func (Callable): returns reward from state, action, next_state + """ + + __metaclass__ = ABCMeta + + def __init__(self, reward_func): + """Initialization. + + Args: + reward_func (Callable): returns reward from state, action, next_state + """ + self.reward_func = reward_func() + + @abstractmethod + def fetch_desired_states_from_demo(self, demo): + pass + + @abstractmethod + def get_desired_state(self, *args): + pass + + @abstractmethod + def generate_demo_transitions(self, demo): + pass + + @abstractmethod + def _get_final_state(self, transition): + pass + + def _append_origin_transitions(self, origin_transitions, transition, desired_state): + """Append original transitions adding goal state for training.""" + origin_transitions.append(self._get_transition(transition, desired_state)) + + def _append_new_transitions(self, new_transitions, transition, final_state): + """Append new transitions made by HER strategy (final) for training.""" + new_transitions.append(self._get_transition(transition, final_state)) + + def _get_transition(self, transition, goal_state): + """Get a single transition concatenated with a goal state.""" + state, action, _, next_state, done = transition + + done = np.array_equal(next_state, goal_state) + reward = self.reward_func(transition, goal_state) + state = np.concatenate((state, goal_state), axis=-1) + next_state = np.concatenate((next_state, goal_state), axis=-1) + + return state, action, reward, next_state, done + + def generate_transitions( + self, transitions, desired_state, success_score, is_demo=False + ): + """Generate new transitions concatenated with desired states.""" + origin_transitions = list() + new_transitions = list() + final_state = self._get_final_state(transitions[-1]) + score = np.sum(np.array(transitions), axis=0)[2] + + for transition in transitions: + # process transitions with the initial goal state + self._append_origin_transitions( + origin_transitions, transition, desired_state + ) + + # do not need to append new transitions if sum of reward is big enough + if not is_demo and score <= success_score: + self._append_new_transitions(new_transitions, transition, final_state) + + return origin_transitions + new_transitions diff --git a/scripts/algorithms/common/abstract/reward_fn.py b/scripts/algorithms/common/abstract/reward_fn.py new file mode 100644 index 0000000..3306d79 --- /dev/null +++ b/scripts/algorithms/common/abstract/reward_fn.py @@ -0,0 +1,19 @@ +# -*- coding: utf-8 -*- +"""Abstract class for computing reward. +- Author: Kh Kim +- Contact: kh.kim@medipixel.io +""" + +from abc import ABCMeta, abstractmethod + + +class RewardFn(object): + """Abstract class for computing reward. + New compute_reward class should redefine __call__() + """ + + __metaclass__ = ABCMeta + + @abstractmethod + def __call__(self, transition, goal_state): + pass diff --git a/scripts/algorithms/sac/agent.py b/scripts/algorithms/sac/agent.py index 060233e..8267f97 100644 --- a/scripts/algorithms/sac/agent.py +++ b/scripts/algorithms/sac/agent.py @@ -8,6 +8,7 @@ """ import os +import pickle import numpy as np import torch @@ -44,10 +45,11 @@ class Agent(AbstractAgent): total_step (int): total step numbers episode_step (int): step number of the current episode i_episode (int): current episode number + her (HER): hinsight experience replay """ - def __init__(self, env, args, hyper_params, models, optims, target_entropy): + def __init__(self, env, args, hyper_params, models, optims, target_entropy, her): """Initialization. Args: @@ -57,6 +59,7 @@ class Agent(AbstractAgent): models (tuple): models including actor and critic optims (tuple): optimizers for actor and critic target_entropy (float): target entropy for the inequality constraint + her (HER): hinsight experience replay """ AbstractAgent.__init__(self, env, args) @@ -69,6 +72,7 @@ class Agent(AbstractAgent): self.total_step = 0 self.episode_step = 0 self.i_episode = 0 + self.her = her # automatic entropy tuning if self.hyper_params["AUTO_ENTROPY_TUNING"]: @@ -92,6 +96,50 @@ class Agent(AbstractAgent): self.hyper_params["BUFFER_SIZE"], self.hyper_params["BATCH_SIZE"] ) + # HER + if self.hyper_params["USE_HER"]: + # load demo replay memory + with open(self.args.demo_path, "rb") as f: + demo = pickle.load(f) + + if self.hyper_params["DESIRED_STATES_FROM_DEMO"]: + self.her.fetch_desired_states_from_demo(demo) + + self.transitions_epi = list() + self.desired_state = np.zeros((1,)) + demo = self.her.generate_demo_transitions(demo) + + if not self.args.test: + # Replay buffers + self.memory = ReplayBuffer( + self.hyper_params["BUFFER_SIZE"], self.hyper_params["BATCH_SIZE"] + ) + + def _preprocess_state(self, state): + """Preprocess state so that actor selects an action.""" + if self.hyper_params["USE_HER"]: + self.desired_state = self.her.get_desired_state() + state = np.concatenate((state, self.desired_state), axis=-1) + state = torch.FloatTensor(state).to(device) + return state + + def _add_transition_to_memory(self, transition): + """Add 1 step and n step transitions to memory.""" + if self.hyper_params["USE_HER"]: + self.transitions_epi.append(transition) + done = transition[-1] or self.episode_step == self.args.max_episode_steps + if done: + # insert generated transitions if the episode is done + transitions = self.her.generate_transitions( + self.transitions_epi, + self.desired_state, + self.hyper_params["SUCCESS_SCORE"], + ) + self.memory.extend(transitions) + self.transitions_epi = list() + else: + self.memory.add(*transition) + def select_action(self, state): """Select an action from the input space.""" self.curr_state = state @@ -111,11 +159,6 @@ class Agent(AbstractAgent): return selected_action.detach().cpu().numpy() - def _preprocess_state(self, state): - """Preprocess state so that actor selects an action.""" - state = torch.FloatTensor(state).to(device) - return state - def step(self, action): """Take an action and return the response of the env.""" self.total_step += 1 @@ -133,10 +176,6 @@ class Agent(AbstractAgent): return next_state, reward, done - def _add_transition_to_memory(self, transition): - """Add 1 step and n step transitions to memory.""" - self.memory.add(*transition) - def update_model(self, experiences): """Train the model after each episode.""" states, actions, rewards, next_states, dones = experiences @@ -304,6 +343,7 @@ class Agent(AbstractAgent): if self.args.log: wandb.init() wandb.config.update(self.hyper_params) + wandb.config.update(vars(self.args)) wandb.watch([self.actor, self.vf, self.qf_1, self.qf_2], log="parameters") for self.i_episode in range(1, self.args.episode_num + 1): diff --git a/scripts/algorithms/td3/agent.py b/scripts/algorithms/td3/agent.py index 7d53fec..723636d 100644 --- a/scripts/algorithms/td3/agent.py +++ b/scripts/algorithms/td3/agent.py @@ -234,6 +234,7 @@ class Agent(AbstractAgent): if self.args.log: wandb.init() wandb.config.update(self.hyper_params) + wandb.config.update(vars(self.args)) wandb.watch([self.actor, self.critic1, self.critic2], log="parameters") for i_episode in range(1, self.args.episode_num + 1): diff --git a/scripts/config/agent/lunarlander_continuous_v2/sac.py b/scripts/config/agent/lunarlander_continuous_v2/sac.py index eaeb02e..6f0c822 100644 --- a/scripts/config/agent/lunarlander_continuous_v2/sac.py +++ b/scripts/config/agent/lunarlander_continuous_v2/sac.py @@ -8,6 +8,7 @@ import numpy as np import torch import torch.optim as optim +from config.agent.lunarlander_continuous_v2.utils import LunarLanderContinuousHER from algorithms.common.networks.mlp import MLP, FlattenMLP, TanhGaussianDistParams from algorithms.sac.agent import Agent @@ -38,6 +39,10 @@ hyper_params = { "VF_HIDDEN_SIZES": [256, 256], "QF_HIDDEN_SIZES": [256, 256], }, + # HER + "USE_HER": True, + "SUCCESS_SCORE": 250.0, + "DESIRED_STATES_FROM_DEMO": True, } @@ -52,6 +57,9 @@ def get(env, args): state_dim = env.observation_space.shape[0] action_dim = env.action_space.shape[0] + if hyper_params["USE_HER"]: + state_dim *= 2 + hidden_sizes_actor = hyper_params["NETWORK"]["ACTOR_HIDDEN_SIZES"] hidden_sizes_vf = hyper_params["NETWORK"]["VF_HIDDEN_SIZES"] hidden_sizes_qf = hyper_params["NETWORK"]["QF_HIDDEN_SIZES"] @@ -107,5 +115,8 @@ def get(env, args): models = (actor, vf, vf_target, qf_1, qf_2) optims = (actor_optim, vf_optim, qf_1_optim, qf_2_optim) + # HER + her = LunarLanderContinuousHER() if hyper_params["USE_HER"] else None + # create an agent - return Agent(env, args, hyper_params, models, optims, target_entropy) + return Agent(env, args, hyper_params, models, optims, target_entropy, her) diff --git a/scripts/config/agent/lunarlander_continuous_v2/utils.py b/scripts/config/agent/lunarlander_continuous_v2/utils.py new file mode 100644 index 0000000..a6bf839 --- /dev/null +++ b/scripts/config/agent/lunarlander_continuous_v2/utils.py @@ -0,0 +1,67 @@ +# -*- coding: utf-8 -*- +"""Utils for examples on LunarLanderContinuous-v2. +- Author: Kh Kim +- Contact: kh.kim@medipixel.io +""" + +import numpy as np + +from algorithms.common.abstract.her import HER +from algorithms.common.abstract.reward_fn import RewardFn + + +class L1DistanceRewardFn(RewardFn): + def __call__(self, transition, goal_state): + """L1 Distance reward function.""" + next_state = transition[3] + eps = 1e-6 + if np.abs(next_state - goal_state).sum() < eps: + return np.float64(0.0) + else: + return np.float64(-1.0) + + +class LunarLanderContinuousHER(HER): + """HER for LunarLanderContinuous-v2 environment. + Attributes: + demo_goal_indices (np.ndarray): indices about goal of demo list + desired_states (np.ndarray): desired states from demonstration + """ + + def __init__(self, reward_func=L1DistanceRewardFn): + """Initialization.""" + HER.__init__(self, reward_func=reward_func) + + # pylint: disable=attribute-defined-outside-init + def fetch_desired_states_from_demo(self, demo): + """Return desired goal states from demonstration data.""" + np_demo = np.array(demo) + self.demo_goal_indices = np.where(np_demo[:, 4])[0] + self.desired_states = np_demo[self.demo_goal_indices][:, 0] + + def get_desired_state(self, *args): + """Sample one of the desired states.""" + return np.random.choice(self.desired_states, 1).item() + + def _get_final_state(self, transition): + """Get final state from transitions for making HER transitions.""" + return transition[0] + + def generate_demo_transitions(self, demo): + """Return generated demo transitions for HER.""" + new_demo = list() + + # generate demo transitions + prev_idx = 0 + for idx in self.demo_goal_indices: + demo_final_state = self._get_final_state(demo[idx]) + transitions = [demo[i] for i in range(prev_idx, idx + 1)] + prev_idx = idx + 1 + + transitions = self.generate_transitions( + transitions, demo_final_state, 0, is_demo=True + ) + + new_demo.extend(transitions) + + return new_demo diff --git a/scripts/config/environment/open_manipulator.py b/scripts/config/environment/open_manipulator.py index 2da7afb..7d22f43 100755 --- a/scripts/config/environment/open_manipulator.py +++ b/scripts/config/environment/open_manipulator.py @@ -2,7 +2,6 @@ from math import pi from geometry_msgs.msg import Quaternion - config = { "ENV_NAME": "OpenManipulatorReacher", "MAX_EPISODE_STEPS": 100, diff --git a/scripts/envs/open_manipulator/open_manipulator_reacher_env.py b/scripts/envs/open_manipulator/open_manipulator_reacher_env.py index 3525ac3..795962a 100755 --- a/scripts/envs/open_manipulator/open_manipulator_reacher_env.py +++ b/scripts/envs/open_manipulator/open_manipulator_reacher_env.py @@ -1,9 +1,9 @@ #! usr/bin/env python -import numpy as np - import gym +import numpy as np from gym.utils import seeding + from ros_interface import ( OpenManipulatorRosGazeboInterface, OpenManipulatorRosRealInterface, diff --git a/scripts/envs/open_manipulator/ros_interface.py b/scripts/envs/open_manipulator/ros_interface.py index e47c385..ddea996 100755 --- a/scripts/envs/open_manipulator/ros_interface.py +++ b/scripts/envs/open_manipulator/ros_interface.py @@ -6,6 +6,8 @@ from math import cos, sin import gym import numpy as np + +import rospkg # noqa import rospy # noqa import tf # noqa import tf.transformations as tr # noqa @@ -17,8 +19,6 @@ from sensor_msgs.msg import JointState from std_msgs.msg import Float64 from urdf_parser_py.urdf import URDF # noqa -import rospkg # noqa - class OpenManipulatorRosBaseInterface(object): """Open Manipulator Interface based on ROS.""" diff --git a/scripts/gazebo_test_open_manipulator.py b/scripts/gazebo_test_open_manipulator.py index 061caca..5fea141 100755 --- a/scripts/gazebo_test_open_manipulator.py +++ b/scripts/gazebo_test_open_manipulator.py @@ -3,9 +3,9 @@ from math import cos, pi, sin import numpy as np +from config.environment.open_manipulator import config as cfg import rospy -from config.environment.open_manipulator import config as cfg from envs.open_manipulator import OpenManipulatorReacherEnv from geometry_msgs.msg import Pose, Quaternion from open_manipulator_msgs.msg import JointPosition, KinematicsPose diff --git a/scripts/run_open_manipulator_reacher_v0.py b/scripts/run_open_manipulator_reacher_v0.py index d1563ed..2547c01 100755 --- a/scripts/run_open_manipulator_reacher_v0.py +++ b/scripts/run_open_manipulator_reacher_v0.py @@ -10,8 +10,9 @@ import argparse import importlib -import algorithms.common.helper_functions as common_utils from config.environment.open_manipulator import config as env_cfg + +import algorithms.common.helper_functions as common_utils from envs.open_manipulator.open_manipulator_reacher_env import OpenManipulatorReacherEnv # configurations From 1b733a9f4c38985c8a186462d3a2e44ad67d1616 Mon Sep 17 00:00:00 2001 From: Whi Kwon <30768596+whikwon@users.noreply.github.com> Date: Thu, 25 Jul 2019 22:57:08 +0900 Subject: [PATCH 3/3] Fix episode never ending issue (#88) --- .../open_manipulator/open_manipulator_reacher_env.py | 9 ++++----- 1 file changed, 4 insertions(+), 5 deletions(-) diff --git a/scripts/envs/open_manipulator/open_manipulator_reacher_env.py b/scripts/envs/open_manipulator/open_manipulator_reacher_env.py index 795962a..17485b3 100755 --- a/scripts/envs/open_manipulator/open_manipulator_reacher_env.py +++ b/scripts/envs/open_manipulator/open_manipulator_reacher_env.py @@ -78,16 +78,15 @@ class OpenManipulatorReacherEnv(gym.Env): # TODO: Add termination condition # if self.ros_interface.check_for_termination(): # self.done = True - if self.ros_interface.check_for_success(): + if ( + self.ros_interface.check_for_success() + or self.episode_steps == self._max_episode_steps + ): self.done = True self.episode_steps = 0 obs = self.ros_interface.get_observation() - if self.episode_steps == self._max_episode_steps: - self.done = False - self.episode_steps = 0 - return obs, self.reward_rescale_ratio * self.reward, self.done, None def reset(self):