fix free log std param (#964)

This commit is contained in:
Eric Liang
2017-09-11 18:52:48 -07:00
committed by Philipp Moritz
parent 99c8b1f38c
commit e17412a72b
2 changed files with 2 additions and 2 deletions
+1 -1
View File
@@ -32,7 +32,7 @@ class ProximalPolicyLoss(object):
# Do not split the last layer of the value function into
# mean parameters and standard deviation parameters and
# do not make the standard deviations free variables.
vf_config["free_logstd"] = False
vf_config["free_log_std"] = False
with tf.variable_scope("value_function"):
self.value_function = ModelCatalog.get_model(
observations, 1, vf_config).outputs
+1 -1
View File
@@ -68,7 +68,7 @@ docker run --shm-size=10G --memory=10G $DOCKER_SHA \
--env CartPole-v1 \
--alg PPO \
--num-iterations 2 \
--config '{"kl_coeff": 1.0, "num_sgd_iter": 10, "sgd_stepsize": 1e-4, "sgd_batchsize": 64, "timesteps_per_batch": 2000, "num_workers": 1}'
--config '{"kl_coeff": 1.0, "num_sgd_iter": 10, "sgd_stepsize": 1e-4, "sgd_batchsize": 64, "timesteps_per_batch": 2000, "num_workers": 1, "model": {"free_log_std": true}}'
docker run --shm-size=10G --memory=10G $DOCKER_SHA \
python /ray/python/ray/rllib/train.py \