diff --git a/README.md b/README.md
index 960f337..479e083 100644
--- a/README.md
+++ b/README.md
@@ -37,4 +37,4 @@ TODO: Write usage instructions
# Credits
- The walker code is adapted from http://rednuht.org/genetic_walkers/
-- The refinforcement learning code uses neurojs
+- DDPG code from metacar
diff --git a/js/ddpg/ddpg.js b/js/ddpg/ddpg.js
new file mode 100644
index 0000000..4c0eafd
--- /dev/null
+++ b/js/ddpg/ddpg.js
@@ -0,0 +1,275 @@
+const { tf } = require('./tf_import')
+const {copyModel, Actor, Critic, assignAndStd } = require('./models')
+
+function logTfMemory(){
+ let mem = tf.memory();
+ console.log("numBytes:" + mem.numBytes +
+ "\nnumBytesInGPU:" + mem.numBytesInGPU +
+ "\nnumDataBuffers:" + mem.numDataBuffers +
+ "\nnumTensors:" + mem.numTensors);
+}
+
+// This class is called from js/DDPG/ddpg_agent.js
+class DDPG {
+
+ /**
+ * @param config (Object)
+ * @param actor (Actor class)
+ * @param critic (Critic class)
+ * @param memory (Memory class)
+ * @param noise (Noise class)
+ */
+ constructor(actor, critic, memory, noise, config){
+ this.actor = actor;
+ this.critic = critic;
+ this.memory = memory;
+ this.noise = noise;
+ this.config = config;
+ this.tfGamma = tf.scalar(config.gamma);
+
+ // Inputs
+ this.obsInput = tf.input({batchShape: [null, this.config.stateSize]});
+ this.actionInput = tf.input({batchShape: [null, this.config.nbActions]});
+
+ // Randomly Initialize actor network μ(s)
+ this.actor.buildModel(this.obsInput);
+ // Randomly Initialize critic network Q(s, a)
+ this.critic.buildModel(this.obsInput, this.actionInput);
+
+
+ // Define in js/DDPG/models.js
+ // Init target network Q' and μ' with the same weights
+ this.actorTarget = copyModel(this.actor, Actor);
+ this.criticTarget = copyModel(this.critic, Critic);
+ // Perturbed Actor (See parameter space noise Exploration paper)
+ this.perturbedActor = copyModel(this.actor, Actor);
+ //this.adaptivePerturbedActor = copyModel(this.actor, Actor);
+
+ this.setLearningOp();
+ }
+
+ setLearningOp(){
+ this.criticWithActor = (tfState) => {
+ return tf.tidy(() => {
+ const tfAct = this.actor.predict(tfState);
+ return this.critic.predict(tfState, tfAct);
+ });
+ };
+ this.criticTargetWithActorTarget = (tfState) => {
+ return tf.tidy(() => {
+ const tfAct = this.actorTarget.predict(tfState);
+ return this.criticTarget.predict(tfState, tfAct);
+ });
+ };
+
+ this.actorOptimiser = tf.train.adam(this.config.actorLr);
+ this.criticOptimiser = tf.train.adam(this.config.criticLr);
+
+ this.criticWeights = [];
+ for (let w = 0; w < this.critic.model.trainableWeights.length; w++){
+ this.criticWeights.push(this.critic.model.trainableWeights[w].val);
+ }
+ this.actorWeights = [];
+ for (let w = 0; w < this.actor.model.trainableWeights.length; w++){
+ this.actorWeights.push(this.actor.model.trainableWeights[w].val);
+ }
+
+ assignAndStd(this.actor, this.perturbedActor, this.noise.currentStddev, this.config.seed);
+ }
+
+
+ /**
+ * Distance Measure for DDPG
+ * See parameter space noise Exploration paper
+ * @param observations (Tensor2d) Observations
+ */
+ distanceMeasure(observations) {
+ return tf.tidy(() => {
+ const pertubedPredictions = this.perturbedActor.model.predict(observations);
+ const predictions = this.actor.model.predict(observations);
+
+ const distance = tf.square(pertubedPredictions.sub(predictions)).mean().sqrt();
+ return distance;
+ });
+ }
+
+ /**
+ * AdaptParamNoise
+ */
+ adaptParamNoise(){
+ const batch = this.memory.getBatch(this.config.batchSize);
+ if (batch.obs0.length == 0){
+ assignAndStd(this.actor, this.perturbedActor, this.noise.currentStddev, this.config.seed);
+ return [0];
+ }
+
+ let distanceV = null;
+
+ if (batch.obs0.length > 0){
+ const tfObs0 = tf.tensor2d(batch.obs0);
+ const distance = this.distanceMeasure(tfObs0);
+
+ assignAndStd(this.actor, this.perturbedActor, this.noise.currentStddev, this.config.seed);
+
+ distanceV = distance.buffer().values;
+ this.noise.adapt(distanceV[0]);
+
+ distance.dispose();
+ tfObs0.dispose();
+ }
+
+ return distanceV;
+ }
+
+ /**
+ * Get the estimation of the Q value given the state
+ * and the action
+ * @param state number[]
+ * @param action [a, steering]
+ */
+ getQvalue(state, a){
+ const st = tf.tensor2d([state]);
+ const tfa = tf.tensor2d([a]);
+ const q = this.critic.model.predict([st, tfa]);
+ const v = q.buffer().values
+ st.dispose();
+ tfa.dispose();
+ q.dispose();
+ return v[0];
+ }
+
+ /**
+ * @param observation (tf.tensor2d)
+ * @return (tf.tensor1d)
+ */
+ predict(observation){
+ const tfActions = this.actor.model.predict(observation);
+ return tfActions;
+ }
+
+ /**
+ * @param observation (tf.tensor2d)
+ * @return (tf.tensor1d)
+ */
+ perturbedPrediction(observation){
+ const tfActions = this.perturbedActor.model.predict(observation);
+ return tfActions;
+ }
+
+ /**
+ * Update the two target network
+ */
+ targetUpdate(){
+ // Define in js/DDPG/models.js
+ targetUpdate(this.criticTarget, this.critic, this.config);
+ targetUpdate(this.actorTarget, this.actor, this.config);
+ }
+
+ trainCritic(batch, tfActions, tfObs0, tfObs1, tfRewards, tfTerminals){
+
+ let costs;
+
+ const criticLoss = this.criticOptimiser.minimize(() => {
+ const tfQPredictions0 = this.critic.model.predict([tfObs0, tfActions]);
+ const tfQPredictions1 = this.criticTargetWithActorTarget(tfObs1);
+
+ const tfQTargets = tfRewards.add(tf.scalar(1).sub(tfTerminals).mul(this.tfGamma).mul(tfQPredictions1));
+
+ const erros = tf.sub(tfQTargets, tfQPredictions0).square();
+ costs = erros.buffer().values;
+
+ return erros.mean();
+ }, true, this.criticWeights);
+
+ // For experience Replay
+ this.memory.appendBackWithCost(batch, costs);
+
+ const loss = criticLoss.buffer().values[0];
+ criticLoss.dispose();
+
+ targetUpdate(this.criticTarget, this.critic, this.config);
+
+ return loss;
+ }
+
+ trainActor(tfObs0){
+
+ const actorLoss = this.actorOptimiser.minimize(() => {
+ const tfQPredictions0 = this.criticWithActor(tfObs0);
+ return tf.mean(tfQPredictions0).mul(tf.scalar(-1.))
+ }, true, this.actorWeights);
+
+ targetUpdate(this.actorTarget, this.actor, this.config);
+
+ const loss = actorLoss.buffer().values[0];
+ actorLoss.dispose();
+
+ return loss;
+ }
+
+ getTfBatch(){
+ // Get batch
+ const batch = this.memory.popBatch(this.config.batchSize);
+ // Convert to tensors
+ const tfActions = tf.tensor2d(batch.actions);
+ const tfObs0 = tf.tensor2d(batch.obs0);
+ const tfObs1 = tf.tensor2d(batch.obs1);
+ const _tfRewards = tf.tensor1d(batch.rewards);
+ const _tfTerminals = tf.tensor1d(batch.terminals);
+
+ const tfRewards = _tfRewards.expandDims(1);
+ const tfTerminals = _tfTerminals.expandDims(1);
+
+ _tfRewards.dispose();
+ _tfTerminals.dispose();
+
+ return {
+ batch, tfActions, tfObs0, tfObs1, tfRewards, tfTerminals
+ }
+ }
+
+ optimizeCritic(){
+ const {batch, tfActions, tfObs0, tfObs1, tfRewards, tfTerminals} = this.getTfBatch();
+
+ const loss = this.trainCritic(tfActions, tfObs0, tfObs1, tfRewards, tfTerminals);
+
+ tfActions.dispose();
+ tfObs0.dispose();
+ tfObs1.dispose();
+ tfRewards.dispose();
+ tfTerminals.dispose();
+
+ return loss;
+ }
+
+ optimizeActor(){
+ const {batch, tfActions, tfObs0, tfObs1, tfRewards, tfTerminals} = this.getTfBatch();
+
+ const loss = this.trainActor(tfObs0);
+
+ tfActions.dispose();
+ tfObs0.dispose();
+ tfObs1.dispose();
+ tfRewards.dispose();
+ tfTerminals.dispose();
+
+ return loss;
+ }
+
+ optimizeCriticActor(){
+ const {batch, tfActions, tfObs0, tfObs1, tfRewards, tfTerminals} = this.getTfBatch();
+
+ const lossC = this.trainCritic(batch, tfActions, tfObs0, tfObs1, tfRewards, tfTerminals);
+ const lossA = this.trainActor(tfObs0);
+
+ tfActions.dispose();
+ tfObs0.dispose();
+ tfObs1.dispose();
+ tfRewards.dispose();
+ tfTerminals.dispose();
+
+ return {lossC, lossA};
+ }
+}
+
+module.exports = { logTfMemory, DDPG }
diff --git a/js/ddpg/ddpg_agent.js b/js/ddpg/ddpg_agent.js
new file mode 100644
index 0000000..bda770b
--- /dev/null
+++ b/js/ddpg/ddpg_agent.js
@@ -0,0 +1,241 @@
+const { tf } = require('./tf_import')
+const AdaptiveParamNoiseSpec = require('./noise')
+const PrioritizedMemory = require('./prioritized_memory')
+const { Actor, Critic, } = require('./models')
+const { DDPG, logTfMemory } = require('./ddpg')
+
+// This class is called from js/DDPG/index.js
+class DDPGAgent {
+
+ /**
+ * @param env (metacar.env) Set in js/DDPG/index.js
+ */
+ constructor(env, config){
+
+ this.stopTraining = false;
+ this.env = env;
+
+ config = config || {};
+ // Default Config
+ this.config = {
+ "stateSize": config.stateSize || 17,
+ "nbActions": config.nbActions || 2,
+ "seed": config.seed || 0,
+ "batchSize": config.batchSize || 128,
+ "actorLr": config.actorLr || 0.0001,
+ "criticLr": config .criticLr || 0.001,
+ "memorySize": config.memorySize || 30000,
+ "gamma": config.gamme || 0.99,
+ "noiseDecay": config.noiseDecay || 0.99,
+ "rewardScale": config.rewardScale || 1,
+ "nbEpochs": config.nbEpochs || 200,
+ "nbEpochsCycle": config.nbEpochsCycle || 10,
+ "nbTrainSteps": config.nbTrainSteps || 110,
+ "tau": config.tau || 0.008,
+ "initialStddev": config.initialStddev || 0.1,
+ "desiredActionStddev": config.desiredActionStddev || 0.1,
+ "adoptionCoefficient": config.adoptionCoefficient || 1.01,
+ "actorFirstLayerSize": config.actorFirstLayerSize || 64,
+ "actorSecondLayerSize": config.actorSecondLayerSize || 32,
+ "criticFirstLayerSSize": config.criticFirstLayerSSize || 64,
+ "criticFirstLayerASize": config.criticFirstLayerASize || 64,
+ "criticSecondLayerSize": config.criticSecondLayerSize || 32,
+ "maxStep": config.maxStep || 800,
+ "stopOnRewardError": config.stopOnRewardError != undefined ? config.stopOnRewardError:true,
+ "resetEpisode": config.resetEpisode != undefined ? config.resetEpisode:false,
+ "saveDuringTraining": config.saveDuringTraining || false,
+ "saveInterval": config.saveInterval || 20
+ };
+ this.epoch = 0;
+ // From js/DDPG/noise.js
+ this.noise = new AdaptiveParamNoiseSpec(this.config);
+
+ // Configure components.
+
+ // Buffer replay
+ // The baseline use 1e6 but this size should be enough for this problem
+ this.memory = new PrioritizedMemory(this.config.memorySize);
+ // Actor and Critic are from js/DDPG/models.js
+ this.actor = new Actor(this.config);
+ this.critic = new Critic(this.config);
+
+ // Seed javascript
+ // Math.seedrandom(0);
+
+ this.rewardsList = [];
+ this.epiDuration = [];
+
+ // DDPG
+ this.ddpg = new DDPG(this.actor, this.critic, this.memory, this.noise, this.config);
+ }
+
+ save(name){
+ /*
+ Save the network
+ */
+ this.ddpg.critic.model.save('downloads://critic-' + name);
+ this.ddpg.actor.model.save('downloads://actor-'+ name);
+ }
+
+ async restore(folder, name){
+ /*
+ Restore the weights of the network
+ */
+ const critic = await tf.loadModel('https://metacar-project.com/public/models/'+folder+'/critic-'+name+'.json');
+ const actor = await tf.loadModel("https://metacar-project.com/public/models/"+folder+"/actor-"+name+".json");
+
+ this.ddpg.critic = copyFromSave(critic, Critic, this.config, this.ddpg.obsInput, this.ddpg.actionInput);
+ this.ddpg.actor = copyFromSave(actor, Actor, this.config, this.ddpg.obsInput, this.ddpg.actionInput);
+
+ // Define in js/DDPG/models.js
+ // Init target network Q' and μ' with the same weights
+ this.ddpg.actorTarget = copyModel(this.ddpg.actor, Actor);
+ this.ddpg.criticTarget = copyModel(this.ddpg.critic, Critic);
+ // Perturbed Actor (See parameter space noise Exploration paper)
+ this.ddpg.perturbedActor = copyModel(this.ddpg.actor, Actor);
+ //this.adaptivePerturbedActor = copyModel(this.actor, Actor);
+ this.ddpg.setLearningOp();
+ }
+
+ /**
+ * Play one step
+ */
+ play(){
+ // Get the current state
+ const state = this.env.getState();
+ // Pick an action
+ const tfActions = this.ddpg.predict(tf.tensor2d([state]));
+ const actions = tfActions.buffer().values;
+ agent.env.step(actions);
+ tfActions.dispose();
+ }
+
+ /**
+ * Get the estimation of the Q value given the state
+ * and the action
+ * @param state number[]
+ * @param action [a, steering]
+ */
+ getQvalue(state, a){
+ return this.ddpg.getQvalue(state, a);
+ }
+
+ stop(){
+ this.stopTraining = true;
+ }
+
+ /**
+ * Step into the training environement
+ * @param tfPreviousStep (tf.tensor2d) Current state
+ * @param mPreviousStep number[]
+ * @return {done, state} One boolean and the new state
+ */
+ stepTrain(tfPreviousStep, mPreviousStep){
+ // Get actions
+ const tfActions = this.ddpg.perturbedPrediction(tfPreviousStep);
+ // Step in the environment with theses actions
+ let mAcions = tfActions.buffer().values;
+ let mReward = this.env.step(mAcions);
+ this.rewardsList.push(mReward);
+ // Get the new observations
+ let mState = this.env.getState();
+ let tfState = tf.tensor2d([mState]);
+ let mDone = 0;
+ if (mReward == -1 && this.config.stopOnRewardError){
+ mDone = 1;
+ }
+ // Add the new tuple to the buffer
+ this.ddpg.memory.append(mPreviousStep, mAcions, mReward, mState, mDone);
+ // Dispose tensors
+ tfPreviousStep.dispose();
+ tfActions.dispose();
+
+ return {mDone, mState, tfState};
+ }
+
+ /**
+ * Optimize models and log states
+ */
+ _optimize(){
+ this.ddpg.noise.desiredActionStddev = Math.max(0.1, this.config.noiseDecay * this.ddpg.noise.desiredActionStddev);
+ let lossValuesCritic = [];
+ let lossValuesActor = [];
+ console.time("Training");
+ for (let t=0; t < this.config.nbTrainSteps; t++){
+ let {lossC, lossA} = this.ddpg.optimizeCriticActor();
+ lossValuesCritic.push(lossC);
+ lossValuesActor.push(lossA);
+ }
+ console.timeEnd("Training");
+ console.log("desiredActionStddev:", this.ddpg.noise.desiredActionStddev);
+ setMetric("CriticLoss", mean(lossValuesCritic));
+ setMetric("ActorLoss", mean(lossValuesActor));
+ }
+
+ /**
+ * Train DDPG Agent
+ */
+ async train(realTime){
+ this.stopTraining = false;
+ // One epoch
+ for (this.epoch; this.epoch < this.config.nbEpochs; this.epoch++){
+ // Perform cycles.
+ this.rewardsList = [];
+ this.stepList = [];
+ this.distanceList = [];
+ // document.getElementById("trainingProgress").innerHTML = "Progression: "+this.epoch+"/"+this.config.nbEpochs+"
";
+ console.log("Progression: "+this.epoch+"/"+this.config.nbEpochs+" epochs")
+ for (let c=0; c < this.config.nbEpochsCycle; c++){
+ if (c%10==0){ logTfMemory(); }
+ let mPreviousStep = this.env.getState();
+ let tfPreviousStep = tf.tensor2d([mPreviousStep]);
+ let step = 0;
+
+ console.time("LoopTime");
+ for (step=0; step < this.config.maxStep; step++){
+ let rel = this.stepTrain(tfPreviousStep, mPreviousStep);
+ mPreviousStep = rel.mState;
+ tfPreviousStep = rel.tfState;
+ if (rel.mDone && this.config.stopOnRewardError){
+ break;
+ }
+ if (this.stopTraining){
+ this.env.render(true);
+ return;
+ }
+ if (realTime && step % 10 == 0)
+ await tf.nextFrame();
+ }
+ this.stepList.push(step);
+ console.timeEnd("LoopTime");
+ let distance = this.ddpg.adaptParamNoise();
+ this.distanceList.push(distance[0]);
+
+ if (this.config.resetEpisode){
+ this.env.reset();
+ }
+ this.env.shuffle({cars: false});
+ tfPreviousStep.dispose();
+ console.log("e="+ this.epoch +", c="+c);
+
+ await tf.nextFrame();
+ }
+ if (this.epoch > 5){
+ this._optimize();
+ }
+ if (this.config.saveDuringTraining && this.epoch % this.config.saveInterval == 0 && this.epoch != 0){
+ this.save("model-ddpg-traffic-epoch-"+this.epoch);
+ }
+ setMetric("Reward", mean(this.rewardsList));
+ setMetric("EpisodeDuration", mean(this.stepList));
+ setMetric("NoiseDistance", mean(this.distanceList));
+ await tf.nextFrame();
+ }
+
+
+ this.env.render(true);
+ }
+
+};
+
+module.exports = DDPGAgent
diff --git a/js/ddpg/models.js b/js/ddpg/models.js
new file mode 100644
index 0000000..407f1dd
--- /dev/null
+++ b/js/ddpg/models.js
@@ -0,0 +1,249 @@
+const { tf } = require('./tf_import')
+
+/**
+ * Copy a model
+ * @param model Actor|Critic instance
+ * @param instance Actor|Critic
+ * @return Copy of the model
+ */
+function copyFromSave(model, instance, config, obs, action){
+ return tf.tidy(() => {
+ nModel = new instance(config);
+ // action might be not required
+ nModel.buildModel(obs, action);
+ const weights = model.weights;
+ for (let m=0; m < weights.length; m++){
+ nModel.model.weights[m].val.assign(weights[m].val);
+ }
+ return nModel;
+ })
+}
+
+
+/**
+ * Copy a model
+ * @param model Actor|Critic instance
+ * @param instance Actor|Critic
+ * @return Copy of the model
+ */
+function copyModel(model, instance){
+ return tf.tidy(() => {
+ nModel = new instance(model.config);
+ // action might be not required
+ nModel.buildModel(model.obs, model.action);
+ const weights = model.model.weights;
+ for (let m=0; m < weights.length; m++){
+ nModel.model.weights[m].val.assign(weights[m].val);
+ }
+ return nModel;
+ })
+}
+
+/**
+ * Copy the value of of the model into the perturbedModel and
+ * add a random pertubation
+ * @param model Actor|Critic instance
+ * @param perturbedActor Actor|Critic instance
+ * @param stddev (number)
+ * @return Copy of the model
+ */
+function assignAndStd(model, perturbedModel, stddev, seed){
+ return tf.tidy(() => {
+ const weights = model.model.trainableWeights;
+ for (let m=0; m < weights.length; m++){
+ let shape = perturbedModel.model.trainableWeights[m].val.shape;
+ let randomTensor = tf.randomNormal(shape, 0, stddev, "float32", seed);
+ let nValue = weights[m].val.add(randomTensor);
+ perturbedModel.model.trainableWeights[m].val.assign(nValue);
+ }
+ });
+}
+
+/**
+ * Update the target models
+ * @param target Actor|Critic instance
+ * @param perturbedActor Actor|Critic instance
+ * @param config (Object)
+ * @return Copy of the model
+ */
+function targetUpdate(target, original, config){
+ return tf.tidy(() => {
+ const originalW = original.model.trainableWeights;
+ const targetW = target.model.trainableWeights;
+
+ const one = tf.scalar(1);
+ const tau = tf.scalar(config.tau);
+
+ for (let m=0; m < originalW.length; m++){
+ const lastValue = target.model.trainableWeights[m].val.clone();
+ let nValue = tau.mul(originalW[m].val).add(targetW[m].val.mul(one.sub(tau)));
+ target.model.trainableWeights[m].val.assign(nValue);
+ const diff = lastValue.sub(target.model.trainableWeights[m].val).mean().buffer().values;
+ if (diff[0] == 0){
+ console.warn("targetUpdate: Nothing have been changed!")
+ }
+ }
+ });
+}
+
+
+class Actor{
+
+ /**
+ @param config (Object)
+ */
+ constructor(config) {
+ this.stateSize = config.stateSize;
+ this.nbActions = config.nbActions;
+ this.layerNorm = config.layerNorm;
+
+ this.firstLayerSize = config.actorFirstLayerSize;
+ this.secondLayerSize = config.actorSecondLayerSize;
+
+ this.seed = config.seed;
+ this.config = config;
+ this.obs = null;
+ }
+
+ /**
+ *
+ * @param obs tf.input
+ */
+ buildModel(obs){
+ this.obs = obs;
+
+ // First layer
+ this.firstLayer = tf.layers.dense({
+ units: this.firstLayerSize,
+ kernelInitializer: tf.initializers.glorotUniform({seed: this.seed}),
+ activation: 'relu',
+ useBias: true,
+ biasInitializer: "zeros"
+ });
+ // Second layer
+ this.secondLayer = tf.layers.dense({
+ units: this.secondLayerSize,
+ kernelInitializer: tf.initializers.glorotUniform({seed: this.seed}),
+ activation: 'relu',
+ useBias: true,
+ biasInitializer: "zeros"
+ });
+ // Ouput layer
+ this.outputLayer = tf.layers.dense({
+ units: this.nbActions,
+ kernelInitializer: tf.initializers.randomUniform({
+ minval: 0.003, maxval: 0.003, seed: this.seed}),
+ activation: 'tanh',
+ useBias: true,
+ biasInitializer: "zeros"
+ });
+ // Actor prediction
+ this.predict = (tfState) => {
+ return tf.tidy(() => {
+ if (tfState){
+ obs = tfState;
+ }
+
+ let l1 = this.firstLayer.apply(obs);
+ let l2 = this.secondLayer.apply(l1);
+
+ return this.outputLayer.apply(l2);
+ });
+ }
+ const output = this.predict();
+ this.model = tf.model({inputs: obs, outputs: output});
+ }
+};
+
+class Critic {
+
+ /**
+ * @param config (Object)
+ */
+ constructor(config) {
+ this.stateSize = config.stateSize;
+ this.nbActions = config.nbActions;
+ this.layerNorm = config.layerNorm;
+
+ this.firstLayerSSize = config.criticFirstLayerSSize
+ this.firstLayerASize = config.criticFirstLayerASize;
+ this.secondLayerSize = config.criticSecondLayerSize;
+
+ this.seed = config.seed;
+ this.config = config;
+ this.obs = null;
+ this.action = null;
+ }
+
+ /**
+ *
+ * @param obs tf.input
+ * @param action tf.input
+ */
+ buildModel(obs, action){
+ this.obs = obs;
+ this.action = action;
+
+ // Used to merged the two first Layer later.
+ this.add = tf.layers.add();
+
+ // First layer
+ this.firstLayerS = tf.layers.dense({
+ units: this.firstLayerSSize,
+ kernelInitializer: tf.initializers.glorotUniform({seed: this.seed}),
+ activation: 'linear', // relu is add later
+ useBias: true,
+ biasInitializer: "zeros"
+ });
+ // First layer
+ this.firstLayerA = tf.layers.dense({
+ units: this.firstLayerASize,
+ kernelInitializer: tf.initializers.glorotUniform({seed: this.seed}),
+ activation: 'linear', // relu is add later
+ useBias: true,
+ biasInitializer: "zeros"
+ });
+ // Second layer
+ this.secondLayer = tf.layers.dense({
+ units: this.secondLayerSize,
+ kernelInitializer: tf.initializers.glorotUniform({seed: this.seed}),
+ activation: 'relu',
+ useBias: true,
+ biasInitializer: "zeros"
+ });
+
+ // Ouput layer
+ this.outputLayer = tf.layers.dense({
+ units: 1,
+ kernelInitializer: tf.initializers.randomUniform({
+ minval: 0.003, maxval: 0.003, seed: this.seed}),
+ activation: 'linear',
+ useBias: true,
+ biasInitializer: "zeros"
+ });
+
+ // Critic prediction
+ this.predict = (tfState, tfActions) => {
+ return tf.tidy(() => {
+ if (tfState && tfActions){
+ obs = tfState;
+ action = tfActions;
+ }
+
+ let l1A = this.firstLayerA.apply(action);
+ let l1S = this.firstLayerS.apply(obs)
+ // Merged layers
+ let concat = this.add.apply([l1A, l1S])
+
+ let l2 = this.secondLayer.apply(concat);
+
+ return this.outputLayer.apply(l2);
+ });
+ }
+
+ const output = this.predict();
+ this.model = tf.model({inputs: [obs, action], outputs: output});
+ }
+};
+
+module.exports = {Actor, Critic, copyFromSave, copyModel, assignAndStd, targetUpdate}
diff --git a/js/ddpg/noise.js b/js/ddpg/noise.js
new file mode 100644
index 0000000..2b19ab4
--- /dev/null
+++ b/js/ddpg/noise.js
@@ -0,0 +1,43 @@
+/**
+ * Noise class
+ * The original baseline is made of three noise
+ * AdaptiveParamNoiseSpec, ActionNoise and NormalActionNoise
+ * Only AdaptiveParamNoiseSpec is implemented for now
+ * See "C Adapative Scaling" Page 14 in the paper.
+ */
+
+class AdaptiveParamNoiseSpec {
+
+ /**
+ * @param conf Object
+ * conf.initialStddev: 0.1 default // σ
+ * conf.desiredActionStddev: 0.1 default // δ
+ * conf.adoptionCoefficient: 1.01 default // α
+ */
+ constructor(conf){
+ conf = conf || {};
+ this.initialStddev = conf.initialStddev || 0.4;
+ this.desiredActionStddev = conf.desiredActionStddev || 0.4;
+ this.adoptionCoefficient = conf.adoptionCoefficient || 1.01;
+ this.currentStddev = this.initialStddev;
+ }
+
+ /**
+ * The distance from the Adaptive scaling
+ * @param distance number
+ */
+ adapt(distance){
+ // if d(π, _π_) > δ then σ = σ/α
+ if (distance > this.desiredActionStddev){
+ // Decrease σ
+ this.currentStddev /= this.adoptionCoefficient;
+ }
+ else{
+ // σ = σ*α
+ // Increase σ
+ this.currentStddev *= this.adoptionCoefficient;
+ }
+ }
+};
+
+module.exports = AdaptiveParamNoiseSpec
diff --git a/js/ddpg/prioritized_memory.js b/js/ddpg/prioritized_memory.js
new file mode 100644
index 0000000..0979a76
--- /dev/null
+++ b/js/ddpg/prioritized_memory.js
@@ -0,0 +1,242 @@
+
+class PrioritizedMemory {
+
+ /**
+ * @param maxlen (number) Buffer limit
+ */
+ constructor(maxlen){
+ this.maxlen = maxlen;
+ this.buffer = [];
+ this.priorBuffer = [];
+ }
+
+ /**
+ * Sample a batch
+ * @param batchSize (number)
+ * @return batch []
+ */
+ getBatch(batchSize){
+ const batch = {
+ 'obs0': [],
+ 'obs1': [],
+ 'rewards': [],
+ 'actions': [],
+ 'terminals': [],
+ };
+
+ if (batchSize > this.priorBuffer.length){
+ console.warn("The size of the replay buffer is < to the batchSize. Return empty batch.");
+ return batch;
+ }
+
+ for (let b=0; b < batchSize/2; b++){
+ let id = Math.floor(Math.random() * this.priorBuffer.length);
+ batch.obs0.push(this.priorBuffer[id].obs0);
+ batch.obs1.push(this.priorBuffer[id].obs1);
+ batch.rewards.push(this.priorBuffer[id].reward);
+ batch.actions.push(this.priorBuffer[id].action);
+ batch.terminals.push(this.priorBuffer[id].terminal);
+ }
+ return batch
+ }
+
+ _bufferBatch(batchSize){
+ const batch = {
+ 'obs0': [],
+ 'obs1': [],
+ 'rewards': [],
+ 'actions': [],
+ 'terminals': [],
+ };
+
+ for (let b=0; b < batchSize/2; b++){
+ let nElem = this.buffer.pop();
+ batch.obs0.push(nElem.obs0);
+ batch.obs1.push(nElem.obs1);
+ batch.rewards.push(nElem.reward);
+ batch.actions.push(nElem.action);
+ batch.terminals.push(nElem.terminal);
+ }
+
+ for (let b=0; b < batchSize/2; b++){
+ let id = Math.floor(Math.random() * this.buffer.length);
+ batch.obs0.push(this.buffer[id].obs0);
+ batch.obs1.push(this.buffer[id].obs1);
+ batch.rewards.push(this.buffer[id].reward);
+ batch.actions.push(this.buffer[id].action);
+ batch.terminals.push(this.buffer[id].terminal);
+ this.buffer.splice(id, 1);
+ }
+
+ return batch
+ }
+
+ _addRandomBufferBatch(batchSize, batch){
+ for (let b=0; b < batchSize; b++){
+ let id = Math.floor(Math.random() * this.buffer.length);
+ batch.obs0.push(this.buffer[id].obs0);
+ batch.obs1.push(this.buffer[id].obs1);
+ batch.rewards.push(this.buffer[id].reward);
+ batch.actions.push(this.buffer[id].action);
+ batch.terminals.push(this.buffer[id].terminal);
+ this.buffer.splice(id, 1);
+ }
+ return batch
+ }
+
+ /**
+ * Sample a batch
+ * @param batchSize (number)
+ * @return batch []
+ */
+ popBatch(batchSize){
+ let originalBatchSize = batchSize;
+ let priorBufferBatchSize;
+ let bufferBatchSize;
+ if (batchSize % 2 != 0){
+ console.warn("Batch size should be a even.")
+ }
+ if (this.priorBuffer.length < batchSize/2){
+ //console.log("get full batch from buffer");
+ const batch = this._bufferBatch(batchSize);
+ console.assert(batch.obs0.length == batchSize);
+ return batch;
+ }
+ const batch = {
+ 'obs0': [],
+ 'obs1': [],
+ 'rewards': [],
+ 'actions': [],
+ 'terminals': [],
+ };
+ if (batchSize > this.length){
+ console.warn("The size of the replay buffer is < to the batchSize. Return empty batch.");
+ return batch;
+ }
+
+ if (this.buffer.length > 0){
+ //console.log("Get half of prior and other from buffer.");
+ batchSize = batchSize / 2;
+ }
+ else{
+ //console.log("Get all from priorBuffer");
+ }
+
+ for (let b=0; b < batchSize; b++){
+ let id = Math.floor(Math.random() * this.priorBuffer.length);
+ batch.obs0.push(this.priorBuffer[id].obs0);
+ batch.obs1.push(this.priorBuffer[id].obs1);
+ batch.rewards.push(this.priorBuffer[id].reward);
+ batch.actions.push(this.priorBuffer[id].action);
+ batch.terminals.push(this.priorBuffer[id].terminal);
+ this.priorBuffer.splice(id, 1);
+ }
+
+ if (this.buffer.length > 0){
+ this._addRandomBufferBatch(batchSize, batch);
+ }
+ console.assert(batch.obs0.length == originalBatchSize);
+ return batch
+ }
+
+ _insert(element, array) {
+ if (array.length == 0 || element.cost < array[0].cost || array[0].cost == null){
+ array.unshift(element);
+ return array;
+ }
+ array.splice(this._locationOf(element, array) + 1, 0, element);
+ return array;
+ }
+
+ _locationOf(element, array, start, end) {
+ start = start || 0;
+ end = end || array.length;
+
+ var pivot = parseInt(start + (end - start) / 2, 10);
+
+ if (end-start <= 1 || array[pivot] === element) return pivot;
+
+ if (array[pivot].cost != null && array[pivot].cost < element.cost) {
+ return this._locationOf(element, array, pivot, end);
+ } else {
+ return this._locationOf(element, array, start, pivot);
+ }
+ }
+
+ /**
+ * @param batch (Object) from getBatch()
+ * @param cost (number) Cost associated with each row of the batch
+ */
+ appendBackWithCost(batch, costs){
+ for (let b=0; b < batch.obs0.length; b++){
+ if (this.buffer.length == this.maxlen){
+ this.buffer.shift();
+ }
+ this._insert({
+ obs0: batch.obs0[b],
+ action: batch.actions[b],
+ reward: batch.rewards[b],
+ obs1: batch.obs1[b],
+ terminal: batch.terminals[b],
+ cost: costs[b]
+ }, this.buffer);
+ }
+ console.assert(this.buffer.length <= this.maxlen);
+ }
+
+ /**
+ * @param obs0 []
+ * @param action (number)
+ * @param reward (number)
+ * @param obs1 []
+ * @param terminal1 (boolean)
+ */
+ append(obs0, action, reward, obs1, terminal){
+ if (this.priorBuffer.length == this.maxlen){
+ this.priorBuffer.shift();
+ }
+ this.priorBuffer.push({
+ obs0: obs0,
+ action: action,
+ reward: reward,
+ obs1: obs1,
+ terminal: terminal,
+ cost: null
+ });
+ console.assert(this.priorBuffer.length <= this.maxlen);
+ }
+}
+/*
+var mem = new Memory(20000);
+Math.seedrandom(0);
+console.assert(mem.length == 0);
+
+var array = [];
+for (let i=1; i < 40000; i++){
+ mem.append("obs0-"+i, "action-"+i, "reward-"+i, "obs1-"+i, "terminal-"+i);
+}
+
+console.assert(mem.length == 20000);
+console.assert(mem.list[0].obs0 == "obs0-20000");
+console.assert(mem.list[19999].obs0 == "obs0-39999");
+
+let batch = mem.getBatch(32);
+
+console.assert(batch.obs0.length == 32);
+console.assert(mem.length == 20000 - 32);
+
+let costs = [];
+for (i=31; i >= 0; i--){
+ costs.push(i);
+}
+mem.appendBackWithCost(batch, costs);
+
+console.log(mem.list);
+
+/*
+for (let i=1; i < 64; i++){
+ mem.append("obs0-"+i, "action-"+i, "reward-"+i, "obs1-"+i, "terminal-"+i);
+}
+*/
+
+module.exports = PrioritizedMemory
diff --git a/js/ddpg/tf_import.js b/js/ddpg/tf_import.js
new file mode 100644
index 0000000..862ebeb
--- /dev/null
+++ b/js/ddpg/tf_import.js
@@ -0,0 +1,2 @@
+const tf = require('@tensorflow/tfjs')
+module.exports = { tf }
diff --git a/js/draw.js b/js/draw.js
index 9a3a895..b8607e3 100644
--- a/js/draw.js
+++ b/js/draw.js
@@ -1,147 +1,154 @@
-drawInit = function() {
- globals.main_screen = document.getElementById("main_screen");
- globals.ctx = main_screen.getContext("2d");
- resetCamera();
-}
+class Renderer {
+ constructor(config, walker, floor) {
+ this.config = config
+ this.walkers = [walker]
+ this.floor = foor
-resetCamera = function() {
- globals.zoom = config.max_zoom_factor;
- globals.translate_x = 0;
- globals.translate_y = 280;
-}
-
-setFps = function(fps) {
- config.draw_fps = fps;
- if(globals.draw_interval)
- clearInterval(globals.draw_interval);
- if(fps > 0 && config.simulation_fps > 0) {
- globals.draw_interval = setInterval(drawFrame, Math.round(1000/config.draw_fps));
+ this.main_screen = document.getElementById("main_screen");
+ this.ctx = main_screen.getContext("2d");
+ resetCamera();
}
-}
-drawFrame = function() {
- var minmax = getMinMaxDistance();
- globals.target_zoom = Math.min(config.max_zoom_factor, getZoom(minmax.min_x, minmax.max_x + 4, minmax.min_y + 2, minmax.max_y + 2.5));
- globals.zoom += 0.1*(globals.target_zoom - globals.zoom);
- globals.translate_x += 0.1*(1.5-minmax.min_x - globals.translate_x);
- globals.translate_y += 0.3*(minmax.min_y*globals.zoom + 280 - globals.translate_y);
- //globals.translate_y = minmax.max_y*globals.zoom + 150;
- globals.ctx.clearRect(0, 0, globals.main_screen.width, globals.main_screen.height);
- globals.ctx.save();
- globals.ctx.translate(globals.translate_x*globals.zoom, globals.translate_y);
- globals.ctx.scale(globals.zoom, -globals.zoom);
- drawFloor();
- for(var k = config.population_size - 1; k >= 0 ; k--) {
- drawWalker(globals.walkers[k]);
+ resetCamera() {
+ this.zoom = config.max_zoom_factor;
+ this.translate_x = 0;
+ this.translate_y = 280;
}
- globals.ctx.restore();
-}
-drawFloor = function() {
- globals.ctx.strokeStyle = "#444";
- globals.ctx.lineWidth = 1/globals.zoom;
- globals.ctx.beginPath();
- var floor_fixture = globals.floor.GetFixtureList();
- globals.ctx.moveTo(floor_fixture.m_shape.m_vertices[0].x, floor_fixture.m_shape.m_vertices[0].y);
- for(var k = 1; k < floor_fixture.m_shape.m_vertices.length; k++) {
- globals.ctx.lineTo(floor_fixture.m_shape.m_vertices[k].x, floor_fixture.m_shape.m_vertices[k].y);
- }
- globals.ctx.stroke();
-}
-
-drawWalker = function (walker) {
- var hue = walker.hue || 240
- globals.ctx.strokeStyle = "hsl(" + hue + ",100%,0%)";
- globals.ctx.fillStyle = "hsl("+hue+",45%,"+(100-15*walker.health/config.walker_health)+"%)";
- globals.ctx.lineWidth = 1/globals.zoom;
-
- // left legs and arms first
- drawRect(walker.left_leg.lower_leg);
- drawRect(walker.left_leg.upper_leg);
- drawRect(walker.left_arm.upper_arm);
- drawRect(walker.left_arm.lower_arm);
-
- globals.ctx.lineWidth = walker.left_leg.frictionJoint.maxForce? 4/globals.zoom : 1/globals.zoom;
- drawRect(walker.left_leg.foot);
- globals.ctx.lineWidth = 1/globals.zoom;
-
- globals.ctx.lineWidth = walker.left_arm.frictionJoint.maxForce? 4/globals.zoom : 1/globals.zoom;
- drawRect(walker.left_arm.hand);
- globals.ctx.lineWidth = 1/globals.zoom;
-
- // head
- drawRect(walker.head.neck);
- drawRect(walker.head.head);
-
- // torso
- drawRect(walker.torso.lower_torso);
- drawRect(walker.torso.upper_torso);
-
- // right legs and arms
- drawRect(walker.right_leg.upper_leg);
- drawRect(walker.right_leg.lower_leg);
- drawRect(walker.right_arm.upper_arm);
- drawRect(walker.right_arm.lower_arm);
-
- globals.ctx.lineWidth = walker.right_leg.frictionJoint.maxForce? 4/globals.zoom : 1/globals.zoom;
- drawRect(walker.right_leg.foot);
- globals.ctx.lineWidth = 1/globals.zoom;
- globals.ctx.lineWidth = walker.right_arm.frictionJoint.maxForce? 4/globals.zoom : 1/globals.zoom;
- drawRect(walker.right_arm.hand);
- globals.ctx.lineWidth = 1/globals.zoom;
-}
-
-drawRect = function(body) {
- // set strokestyle and fillstyle before call
- globals.ctx.beginPath();
- var fixture = body.GetFixtureList();
- var shape = fixture.GetShape();
- var p0 = body.GetWorldPoint(shape.m_vertices[0]);
- globals.ctx.moveTo(p0.x, p0.y);
- for(var k = 1; k < 4; k++) {
- var p = body.GetWorldPoint(shape.m_vertices[k]);
- globals.ctx.lineTo(p.x, p.y);
- }
- globals.ctx.lineTo(p0.x, p0.y);
-
- globals.ctx.fill();
- globals.ctx.stroke();
-}
-
-drawTest = function() {
- globals.ctx.strokeStyle = "#000";
- globals.ctx.fillStyle = "#666";
- globals.ctx.lineWidth = 1;
- globals.ctx.beginPath();
- globals.ctx.moveTo(0, 0);
- globals.ctx.lineTo(0, 2);
- globals.ctx.lineTo(2, 2);
-
- globals.ctx.fill();
- globals.ctx.stroke();
-
-}
-
-getMinMaxDistance = function() {
- var min_x = 9999;
- var max_x = -1;
- var min_y = 9999;
- var max_y = -1;
- for(var k = 0; k < globals.walkers.length; k++) {
- if(globals.walkers[k].health > 0) {
- var dist = globals.walkers[k].torso.upper_torso.GetPosition();
- min_x = Math.min(min_x, dist.x);
- max_x = Math.max(max_x, dist.x);
- min_y = Math.min(min_y, globals.walkers[k].low_foot_height, globals.walkers[k].head_height);
- max_y = Math.max(max_y, dist.y);
+ setFps(fps) {
+ config.draw_fps = fps;
+ if(this.draw_interval)
+ clearInterval(this.draw_interval);
+ if(fps > 0 && config.simulation_fps > 0) {
+ this.draw_interval = setInterval(drawFrame, Math.round(1000/config.draw_fps));
}
}
- return {min_x:min_x, max_x:max_x, min_y:min_y, max_y:max_y};
-}
-getZoom = function(min_x, max_x, min_y, max_y) {
- var delta_x = Math.abs(max_x - min_x);
- var delta_y = Math.abs(max_y - min_y);
- var zoom = Math.min(globals.main_screen.width/delta_x,globals.main_screen.height/delta_y);
- return zoom;
+ drawFrame() {
+ var minmax = getMinMaxDistance();
+ this.target_zoom = Math.min(config.max_zoom_factor, getZoom(minmax.min_x, minmax.max_x + 4, minmax.min_y + 2, minmax.max_y + 2.5));
+ this.zoom += 0.1*(this.target_zoom - this.zoom);
+ this.translate_x += 0.1*(1.5-minmax.min_x - this.translate_x);
+ this.translate_y += 0.3*(minmax.min_y*this.zoom + 280 - this.translate_y);
+ //this.translate_y = minmax.max_y*this.zoom + 150;
+ this.ctx.clearRect(0, 0, this.main_screen.width, this.main_screen.height);
+ this.ctx.save();
+ this.ctx.translate(this.translate_x*this.zoom, this.translate_y);
+ this.ctx.scale(this.zoom, -this.zoom);
+ drawFloor();
+ for(var k = config.population_size - 1; k >= 0 ; k--) {
+ drawWalker(this.walkers[k]);
+ }
+ this.ctx.restore();
+ }
+
+ drawFloor() {
+ this.ctx.strokeStyle = "#444";
+ this.ctx.lineWidth = 1/this.zoom;
+ this.ctx.beginPath();
+ var floor_fixture = this.floor.GetFixtureList();
+ this.ctx.moveTo(floor_fixture.m_shape.m_vertices[0].x, floor_fixture.m_shape.m_vertices[0].y);
+ for(var k = 1; k < floor_fixture.m_shape.m_vertices.length; k++) {
+ this.ctx.lineTo(floor_fixture.m_shape.m_vertices[k].x, floor_fixture.m_shape.m_vertices[k].y);
+ }
+ this.ctx.stroke();
+ }
+
+ drawWalker (walker) {
+ var hue = walker.hue || 240
+ this.ctx.strokeStyle = "hsl(" + hue + ",100%,0%)";
+ this.ctx.fillStyle = "hsl("+hue+",45%,"+(100-15*walker.health/config.walker_health)+"%)";
+ this.ctx.lineWidth = 1/this.zoom;
+
+ // left legs and arms first
+ drawRect(walker.left_leg.lower_leg);
+ drawRect(walker.left_leg.upper_leg);
+ drawRect(walker.left_arm.upper_arm);
+ drawRect(walker.left_arm.lower_arm);
+
+ this.ctx.lineWidth = walker.left_leg.frictionJoint.maxForce? 4/this.zoom : 1/this.zoom;
+ drawRect(walker.left_leg.foot);
+ this.ctx.lineWidth = 1/this.zoom;
+
+ this.ctx.lineWidth = walker.left_arm.frictionJoint.maxForce? 4/this.zoom : 1/this.zoom;
+ drawRect(walker.left_arm.hand);
+ this.ctx.lineWidth = 1/this.zoom;
+
+ // head
+ drawRect(walker.head.neck);
+ drawRect(walker.head.head);
+
+ // torso
+ drawRect(walker.torso.lower_torso);
+ drawRect(walker.torso.upper_torso);
+
+ // right legs and arms
+ drawRect(walker.right_leg.upper_leg);
+ drawRect(walker.right_leg.lower_leg);
+ drawRect(walker.right_arm.upper_arm);
+ drawRect(walker.right_arm.lower_arm);
+
+ this.ctx.lineWidth = walker.right_leg.frictionJoint.maxForce? 4/this.zoom : 1/this.zoom;
+ drawRect(walker.right_leg.foot);
+ this.ctx.lineWidth = 1/this.zoom;
+ this.ctx.lineWidth = walker.right_arm.frictionJoint.maxForce? 4/this.zoom : 1/this.zoom;
+ drawRect(walker.right_arm.hand);
+ this.ctx.lineWidth = 1/this.zoom;
+ }
+
+ drawRect(body) {
+ // set strokestyle and fillstyle before call
+ this.ctx.beginPath();
+ var fixture = body.GetFixtureList();
+ var shape = fixture.GetShape();
+ var p0 = body.GetWorldPoint(shape.m_vertices[0]);
+ this.ctx.moveTo(p0.x, p0.y);
+ for(var k = 1; k < 4; k++) {
+ var p = body.GetWorldPoint(shape.m_vertices[k]);
+ this.ctx.lineTo(p.x, p.y);
+ }
+ this.ctx.lineTo(p0.x, p0.y);
+
+ this.ctx.fill();
+ this.ctx.stroke();
+ }
+
+ drawTest() {
+ this.ctx.strokeStyle = "#000";
+ this.ctx.fillStyle = "#666";
+ this.ctx.lineWidth = 1;1
+ this.ctx.beginPath();
+ this.ctx.moveTo(0, 0);
+ this.ctx.lineTo(0, 2);
+ this.ctx.lineTo(2, 2);
+
+ this.ctx.fill();
+ this.ctx.stroke();
+
+ }
+
+ getMinMaxDistance() {
+ var min_x = 9999;
+ var max_x = -1;
+ var min_y = 9999;
+ var max_y = -1;
+ for(var k = 0; k < this.walkers.length; k++) {
+ if(this.walkers[k].health > 0) {
+ var dist = this.walkers[k].torso.upper_torso.GetPosition();
+ min_x = Math.min(min_x, dist.x);
+ max_x = Math.max(max_x, dist.x);
+ min_y = Math.min(min_y, this.walkers[k].low_foot_height, this.walkers[k].head_height);
+ max_y = Math.max(max_y, dist.y);
+ }
+ }
+ return {min_x:min_x, max_x:max_x, min_y:min_y, max_y:max_y};
+ }
+
+ getZoom(min_x, max_x, min_y, max_y) {
+ var delta_x = Math.abs(max_x - min_x);
+ var delta_y = Math.abs(max_y - min_y);
+ var zoom = Math.min(this.main_screen.width/delta_x,this.main_screen.height/delta_y);
+ return zoom;
+ }
}
+module.export = {Renderer}
diff --git a/js/floor.js b/js/floor.js
index 3dce45c..503363e 100644
--- a/js/floor.js
+++ b/js/floor.js
@@ -1,4 +1,6 @@
-function createFloor(world) {
+var b2 = require('../vendor/jsbox2d')
+
+function createFloor(world, max_floor_tiles) {
var body_def = new b2.BodyDef();
var body = world.CreateBody(body_def);
body.SetUserData('floor')
@@ -13,8 +15,8 @@ function createFloor(world) {
new b2.Vec2(2.5, -0.16)
];
- for(var k = 2; k < config.max_floor_tiles; k++) {
- var ratio = k / config.max_floor_tiles;
+ for(var k = 2; k < max_floor_tiles; k++) {
+ var ratio = k / max_floor_tiles;
// add uneven floor by continuing from the last point, plus some random jittering
edges.push(new b2.Vec2(
edges[edges.length - 1].x + (1 + ratio * Math.random() - ratio / 2),
@@ -26,3 +28,5 @@ function createFloor(world) {
body.CreateFixture(fix_def);
return body;
}
+
+module.exports=createFloor
diff --git a/js/walker.js b/js/walker.js
index 8dd3468..8386ae9 100644
--- a/js/walker.js
+++ b/js/walker.js
@@ -1,21 +1,22 @@
// walker has fixed shapes and structures
// shape definitions are in the constructor
+var randf = (low, high) => Math.random() * (high - low) + low
-
-function deg2rad(deg) {
- return deg/180*Math.PI
+function deg2rad(deg) {
+ return deg / 180 * Math.PI
}
const STRENGTH = 3
-var Walker = function() {
+var Walker = function () {
this.__constructor.apply(this, arguments);
}
-Walker.prototype.__constructor = function(world, floor) {
+Walker.prototype.__constructor = function (world, floor, config) {
this.world = world;
this.floor = floor
+ this.config = config
this.density = 106.2; // common for all fixtures, no reason to be too specific
@@ -25,13 +26,18 @@ Walker.prototype.__constructor = function(world, floor) {
this.low_foot_height = 0;
this.head_height = 0;
this.steps = 0;
- this.distance = 0
+ this.distance = 0
this.last_left_left_forward = true
- this.hue = Math.randf(200,360)
+ this.hue = randf(200, 360)
- this.bd = new b2.BodyDef({positions: {x:10, y:-10}});
- this.bd.position.x += Math.randf(-10, 10)
+ this.bd = new b2.BodyDef({
+ positions: {
+ x: 10,
+ y: -10
+ }
+ });
+ this.bd.position.x += randf(-10, 10)
this.bd.type = b2.Body.b2_dynamicBody;
this.bd.linearDamping = 0;
this.bd.angularDamping = 20; // decay in force/ air friction
@@ -99,25 +105,24 @@ Walker.prototype.__constructor = function(world, floor) {
// add grip
// we don't have data on external forces, so I will just punish for contact with the ground
- http://blog.sethladd.com/2011/09/box2d-collision-damage-for-javascript.html
- // However we could use a listener or calc force
- var self = this
+ http: //blog.sethladd.com/2011/09/box2d-collision-damage-for-javascript.html
+ // However we could use a listener or calc force
+ var self = this
this.contactListener = new b2.ContactListener()
this.contactListener.BeginContact = function (contact, impulse) {
if (contact.m_fixtureA.m_body.m_userData == "floor" | contact.m_fixtureB.m_body.m_userData) {
var otherFixture = contact.m_fixtureA.m_body.m_userData == "floor" ? contact.m_fixtureB : contact.m_fixtureA
- if (otherFixture.m_body === self.right_leg.foot) {
+ if (otherFixture.m_body === self.right_leg.foot) {
// TODO let the agent act to grip or not. Only if palm or foot down?
self.right_leg.frictionJoint.maxForce = 1000 * self.grips[0]
self.right_leg.frictionJoint.maxTorque = 1000 * self.grips[0]
} else if (otherFixture.m_body === self.left_leg.foot) {
self.left_leg.frictionJoint.maxForce = 1000 * self.grips[1]
self.left_leg.frictionJoint.maxTorque = 1000 * self.grips[1]
- } else if (otherFixture.m_body == self.right_arm.hand) {
- // TODO let the agent act to grip or not
+ } else if (otherFixture.m_body == self.right_arm.hand) {
self.right_arm.frictionJoint.maxForce = 1000 * self.grips[2]
self.right_arm.frictionJoint.maxTorque = 1000 * self.grips[2]
- } else if (otherFixture.m_body === self.left_arm.hand) {
+ } else if (otherFixture.m_body === self.left_arm.hand) {
self.left_arm.frictionJoint.maxForce = 1000 * self.grips[3]
self.left_arm.frictionJoint.maxTorque = 1000 * self.grips[3]
}
@@ -126,29 +131,34 @@ Walker.prototype.__constructor = function(world, floor) {
this.contactListener.EndContact = function (contact, impulse) {
if (contact.m_fixtureA.m_body.m_userData == "floor" | contact.m_fixtureB.m_body.m_userData) {
var otherFixture = contact.m_fixtureA.m_body.m_userData == "floor" ? contact.m_fixtureB : contact.m_fixtureA
- if (otherFixture.m_body.m_userData == "right_foot") {
- // console.log('grip off ' + otherFixture.m_body.m_userData)
- // TODO let the agent act to grip or not
- setTimeout(() => { self.right_leg.frictionJoint.maxForce = 0 }, 100)
- setTimeout(() => { self.right_leg.frictionJoint.maxTorque = 0 }, 100)
- } else if (otherFixture.m_body.m_userData == "left_foot") {
- // console.log('grip off ' + otherFixture.m_body.m_userData)
+ if (otherFixture.m_body === self.right_leg.foot) {
+ setTimeout(() => {
+ self.right_leg.frictionJoint.maxForce = 0
+ }, 100)
+ setTimeout(() => {
+ self.right_leg.frictionJoint.maxTorque = 0
+ }, 100)
+ } else if (otherFixture.m_body === self.left_leg.foot) {
self.left_leg.frictionJoint.maxForce = 0
self.left_leg.frictionJoint.maxTorque = 0
- } else if (otherFixture.m_body.m_userData == "right_hand") {
- // console.log('grip off ' + otherFixture.m_body.m_userData)
- // TODO let the agent act to grip or not
- setTimeout(() => { self.right_arm.frictionJoint.maxForce = 0 }, 100)
- setTimeout(() => { self.right_arm.frictionJoint.maxTorque = 0 }, 100)
- } else if (otherFixture.m_body.m_userData == "left_hand") {
- // console.log('grip off ' + otherFixture.m_body.m_userData)
+ } else if (otherFixture.m_body == self.right_arm.hand) {
+ setTimeout(() => {
+ self.right_arm.frictionJoint.maxForce = 0
+ }, 100)
+ setTimeout(() => {
+ self.right_arm.frictionJoint.maxTorque = 0
+ }, 100)
+ } else if (otherFixture.m_body === self.left_arm.hand) {
self.left_arm.frictionJoint.maxForce = 0
self.left_arm.frictionJoint.maxTorque = 0
}
- }
+ }
}
this.world.SetContactListener(this.contactListener)
+ // if (typeof document!==undefined)
+ // this.renderer = new Renderer()
+
}
// Walker.prototype.destroy = function () {
@@ -158,21 +168,21 @@ Walker.prototype.__constructor = function(world, floor) {
// this.otherJoints.map(joint => this.world.DestroyJoint(joint))
// }
-Walker.prototype.createTorso = function() {
+Walker.prototype.createTorso = function () {
// upper torso
- this.bd.position.Set(0.5 - this.leg_def.foot_length/2 + this.leg_def.tibia_width/2, this.leg_def.foot_height/2 + this.leg_def.foot_height/2 + this.leg_def.tibia_length + this.leg_def.femur_length + this.torso_def.lower_height + this.torso_def.upper_height/2);
+ this.bd.position.Set(0.5 - this.leg_def.foot_length / 2 + this.leg_def.tibia_width / 2, this.leg_def.foot_height / 2 + this.leg_def.foot_height / 2 + this.leg_def.tibia_length + this.leg_def.femur_length + this.torso_def.lower_height + this.torso_def.upper_height / 2);
var upper_torso = this.world.CreateBody(this.bd);
upper_torso.SetUserData('upper_torso')
- this.fd.shape.SetAsBox(this.torso_def.upper_width/2, this.torso_def.upper_height/2);
+ this.fd.shape.SetAsBox(this.torso_def.upper_width / 2, this.torso_def.upper_height / 2);
upper_torso.CreateFixture(this.fd);
// lower torso
- this.bd.position.Set(0.5 - this.leg_def.foot_length/2 + this.leg_def.tibia_width/2, this.leg_def.foot_height/2 + this.leg_def.foot_height/2 + this.leg_def.tibia_length + this.leg_def.femur_length + this.torso_def.lower_height/2);
+ this.bd.position.Set(0.5 - this.leg_def.foot_length / 2 + this.leg_def.tibia_width / 2, this.leg_def.foot_height / 2 + this.leg_def.foot_height / 2 + this.leg_def.tibia_length + this.leg_def.femur_length + this.torso_def.lower_height / 2);
var lower_torso = this.world.CreateBody(this.bd);
- lower_torso.SetUserData('lower_torso' )
+ lower_torso.SetUserData('lower_torso')
- this.fd.shape.SetAsBox(this.torso_def.lower_width/2, this.torso_def.lower_height/2);
+ this.fd.shape.SetAsBox(this.torso_def.lower_width / 2, this.torso_def.lower_height / 2);
lower_torso.CreateFixture(this.fd);
// torso joint
@@ -180,53 +190,56 @@ Walker.prototype.createTorso = function() {
// For 3d definition https://github.com/openai/gym/blob/master/gym/envs/mujoco/assets/humanoid.xml
var jd = new b2.RevoluteJointDef();
var position = upper_torso.GetPosition().Clone();
- position.y -= this.torso_def.upper_height/2;
- position.x -= this.torso_def.lower_width/3;
+ position.y -= this.torso_def.upper_height / 2;
+ position.x -= this.torso_def.lower_width / 3;
jd.Initialize(upper_torso, lower_torso, position);
- jd.lowerAngle = deg2rad(-75/2);
+ jd.lowerAngle = deg2rad(-75 / 2);
jd.upperAngle = deg2rad(30 / 2);
jd.enableLimit = true;
jd.maxMotorTorque = 150 * STRENGTH;
jd.motorSpeed = 0;
jd.enableMotor = true;
var j = this.world.CreateJoint(jd)
- j.SetUserData('torso_joint' )
+ j.SetUserData('torso_joint')
this.joints.push(j);
- return {upper_torso: upper_torso, lower_torso: lower_torso};
+ return {
+ upper_torso: upper_torso,
+ lower_torso: lower_torso
+ };
}
-Walker.prototype.createLeg = function(label) {
+Walker.prototype.createLeg = function (label) {
// upper leg
- this.bd.position.Set(0.5 - this.leg_def.foot_length/2 + this.leg_def.tibia_width/2, this.leg_def.foot_height/2 + this.leg_def.foot_height/2 + this.leg_def.tibia_length + this.leg_def.femur_length/2);
+ this.bd.position.Set(0.5 - this.leg_def.foot_length / 2 + this.leg_def.tibia_width / 2, this.leg_def.foot_height / 2 + this.leg_def.foot_height / 2 + this.leg_def.tibia_length + this.leg_def.femur_length / 2);
var upper_leg = this.world.CreateBody(this.bd);
- this.fd.shape.SetAsBox(this.leg_def.femur_width/2, this.leg_def.femur_length/2);
+ this.fd.shape.SetAsBox(this.leg_def.femur_width / 2, this.leg_def.femur_length / 2);
upper_leg.CreateFixture(this.fd);
upper_leg.SetUserData(label + 'upper_leg')
// lower leg
- this.bd.position.Set(0.5 - this.leg_def.foot_length/2 + this.leg_def.tibia_width/2, this.leg_def.foot_height/2 + this.leg_def.foot_height/2 + this.leg_def.tibia_length/2);
+ this.bd.position.Set(0.5 - this.leg_def.foot_length / 2 + this.leg_def.tibia_width / 2, this.leg_def.foot_height / 2 + this.leg_def.foot_height / 2 + this.leg_def.tibia_length / 2);
var lower_leg = this.world.CreateBody(this.bd);
- this.fd.shape.SetAsBox(this.leg_def.tibia_width/2, this.leg_def.tibia_length/2);
+ this.fd.shape.SetAsBox(this.leg_def.tibia_width / 2, this.leg_def.tibia_length / 2);
lower_leg.CreateFixture(this.fd);
lower_leg.SetUserData(label + 'lower_leg')
// foot
- this.bd.position.Set(0.5, this.leg_def.foot_height/2);
+ this.bd.position.Set(0.5, this.leg_def.foot_height / 2);
var foot = this.world.CreateBody(this.bd);
- this.fd.shape.SetAsBox(this.leg_def.foot_length/2, this.leg_def.foot_height/2);
+ this.fd.shape.SetAsBox(this.leg_def.foot_length / 2, this.leg_def.foot_height / 2);
foot.CreateFixture(this.fd);
foot.SetUserData(label + 'foot')
var fjd = new b2.FrictionJointDef();
- var position = new b2.Vec2(0,0)
+ var position = new b2.Vec2(0, 0)
fjd.Initialize(foot, this.floor, position)
fjd.maxForce = 0; //This the most force the joint will apply to your object. The faster its moving the more force applied
fjd.maxTorque = 0; //Set to 0 to prevent rotation
- fjd.userData = label+'foot_friction_joint'
+ fjd.userData = label + 'foot_friction_joint'
fjd.collideConnected = true
var frictionJoint = this.world.CreateJoint(fjd)
this.otherJoints.push(frictionJoint)
@@ -234,8 +247,8 @@ Walker.prototype.createLeg = function(label) {
// leg joints
var jd = new b2.RevoluteJointDef();
var position = upper_leg.GetPosition().Clone();
- position.y -= this.leg_def.femur_length/2;
- position.x += this.leg_def.femur_width/4;
+ position.y -= this.leg_def.femur_length / 2;
+ position.x += this.leg_def.femur_width / 4;
jd.Initialize(upper_leg, lower_leg, position);
jd.lowerAngle = deg2rad(-100);
jd.upperAngle = deg2rad(-2);
@@ -250,7 +263,7 @@ Walker.prototype.createLeg = function(label) {
// foot joint
var jd = new b2.RevoluteJointDef();
var position = lower_leg.GetPosition().Clone();
- position.y -= this.leg_def.tibia_length/2;
+ position.y -= this.leg_def.tibia_length / 2;
jd.Initialize(lower_leg, foot, position);
jd.lowerAngle = -deg2rad(-36);
jd.upperAngle = deg2rad(30);
@@ -262,41 +275,46 @@ Walker.prototype.createLeg = function(label) {
j.SetUserData(label + 'foot_joint')
this.joints.push(j);
- return {upper_leg: upper_leg, lower_leg: lower_leg, foot:foot, frictionJoint:frictionJoint};
+ return {
+ upper_leg: upper_leg,
+ lower_leg: lower_leg,
+ foot: foot,
+ frictionJoint: frictionJoint
+ };
}
-Walker.prototype.createArm = function(label) {
+Walker.prototype.createArm = function (label) {
// upper arm
- this.bd.position.Set(0.5 - this.leg_def.foot_length/2 + this.leg_def.tibia_width/2, this.leg_def.foot_height/2 + this.leg_def.foot_height/2 + this.leg_def.tibia_length + this.leg_def.femur_length + this.torso_def.lower_height + this.torso_def.upper_height - this.arm_def.arm_length/2);
+ this.bd.position.Set(0.5 - this.leg_def.foot_length / 2 + this.leg_def.tibia_width / 2, this.leg_def.foot_height / 2 + this.leg_def.foot_height / 2 + this.leg_def.tibia_length + this.leg_def.femur_length + this.torso_def.lower_height + this.torso_def.upper_height - this.arm_def.arm_length / 2);
var upper_arm = this.world.CreateBody(this.bd);
- this.fd.shape.SetAsBox(this.arm_def.arm_width/2, this.arm_def.arm_length/2);
+ this.fd.shape.SetAsBox(this.arm_def.arm_width / 2, this.arm_def.arm_length / 2);
upper_arm.CreateFixture(this.fd);
upper_arm.SetUserData(label + 'upper_arm')
// lower arm
- this.bd.position.Set(0.5 - this.leg_def.foot_length/2 + this.leg_def.tibia_width/2, this.leg_def.foot_height/2 + this.leg_def.foot_height/2 + this.leg_def.tibia_length + this.leg_def.femur_length + this.torso_def.lower_height + this.torso_def.upper_height - this.arm_def.arm_length - this.arm_def.forearm_length/2);
+ this.bd.position.Set(0.5 - this.leg_def.foot_length / 2 + this.leg_def.tibia_width / 2, this.leg_def.foot_height / 2 + this.leg_def.foot_height / 2 + this.leg_def.tibia_length + this.leg_def.femur_length + this.torso_def.lower_height + this.torso_def.upper_height - this.arm_def.arm_length - this.arm_def.forearm_length / 2);
var lower_arm = this.world.CreateBody(this.bd);
- this.fd.shape.SetAsBox(this.arm_def.forearm_width/2, this.arm_def.forearm_length/2);
+ this.fd.shape.SetAsBox(this.arm_def.forearm_width / 2, this.arm_def.forearm_length / 2);
lower_arm.CreateFixture(this.fd);
lower_arm.SetUserData(label + 'lower_arm')
// hand
- this.bd.position.Set(0.5 - this.leg_def.foot_length/2 + this.leg_def.tibia_width/2, this.leg_def.foot_height/2 + this.leg_def.foot_height/2 + this.leg_def.tibia_length + this.leg_def.femur_length + this.torso_def.lower_height + this.torso_def.upper_height - this.arm_def.arm_length - this.arm_def.forearm_length - this.arm_def.hand_length/2);
+ this.bd.position.Set(0.5 - this.leg_def.foot_length / 2 + this.leg_def.tibia_width / 2, this.leg_def.foot_height / 2 + this.leg_def.foot_height / 2 + this.leg_def.tibia_length + this.leg_def.femur_length + this.torso_def.lower_height + this.torso_def.upper_height - this.arm_def.arm_length - this.arm_def.forearm_length - this.arm_def.hand_length / 2);
var hand = this.world.CreateBody(this.bd);
- this.fd.shape.SetAsBox(this.arm_def.hand_height/2, this.arm_def.hand_length/2);
+ this.fd.shape.SetAsBox(this.arm_def.hand_height / 2, this.arm_def.hand_length / 2);
hand.CreateFixture(this.fd);
hand.SetUserData(label + 'hand')
var fjd = new b2.FrictionJointDef();
- var position = new b2.Vec2(0,0)
+ var position = new b2.Vec2(0, 0)
fjd.Initialize(hand, this.floor, position)
fjd.maxForce = 0; //This the most force the joint will apply to your object. The faster its moving the more force applied
fjd.maxTorque = 0; //Set to 0 to prevent rotation
- fjd.userData = label+'hand_friction_joint'
+ fjd.userData = label + 'hand_friction_joint'
fjd.collideConnected = true
var frictionJoint = this.world.CreateJoint(fjd)
this.otherJoints.push(frictionJoint)
@@ -305,7 +323,7 @@ Walker.prototype.createArm = function(label) {
// arm join
var jd = new b2.RevoluteJointDef();
var position = upper_arm.GetPosition().Clone();
- position.y -= this.arm_def.arm_length/2;
+ position.y -= this.arm_def.arm_length / 2;
jd.Initialize(upper_arm, lower_arm, position);
jd.lowerAngle = deg2rad(0);
jd.upperAngle = deg2rad(85);
@@ -320,7 +338,7 @@ Walker.prototype.createArm = function(label) {
// hand joint
var jd = new b2.RevoluteJointDef();
var position = lower_arm.GetPosition().Clone();
- position.y -= this.arm_def.forearm_length/2;
+ position.y -= this.arm_def.forearm_length / 2;
jd.Initialize(lower_arm, hand, position);
jd.lowerAngle = deg2rad(-35);
jd.upperAngle = deg2rad(35);
@@ -333,33 +351,34 @@ Walker.prototype.createArm = function(label) {
this.joints.push(j);
return {
- upper_arm: upper_arm, lower_arm: lower_arm,
+ upper_arm: upper_arm,
+ lower_arm: lower_arm,
hand: hand,
- frictionJoint:frictionJoint
+ frictionJoint: frictionJoint
};
}
-Walker.prototype.createHead = function() {
+Walker.prototype.createHead = function () {
// neck
- this.bd.position.Set(0.5 - this.leg_def.foot_length/2 + this.leg_def.tibia_width/2, this.leg_def.foot_height/2 + this.leg_def.foot_height/2 + this.leg_def.tibia_length + this.leg_def.femur_length + this.torso_def.lower_height + this.torso_def.upper_height + this.head_def.neck_height/2);
+ this.bd.position.Set(0.5 - this.leg_def.foot_length / 2 + this.leg_def.tibia_width / 2, this.leg_def.foot_height / 2 + this.leg_def.foot_height / 2 + this.leg_def.tibia_length + this.leg_def.femur_length + this.torso_def.lower_height + this.torso_def.upper_height + this.head_def.neck_height / 2);
var neck = this.world.CreateBody(this.bd);
-
- this.fd.shape.SetAsBox(this.head_def.neck_width/2, this.head_def.neck_height/2);
+
+ this.fd.shape.SetAsBox(this.head_def.neck_width / 2, this.head_def.neck_height / 2);
neck.CreateFixture(this.fd);
neck.SetUserData('neck')
// head
- this.bd.position.Set(0.5 - this.leg_def.foot_length/2 + this.leg_def.tibia_width/2, this.leg_def.foot_height/2 + this.leg_def.foot_height/2 + this.leg_def.tibia_length + this.leg_def.femur_length + this.torso_def.lower_height + this.torso_def.upper_height + this.head_def.neck_height + this.head_def.head_height/2);
+ this.bd.position.Set(0.5 - this.leg_def.foot_length / 2 + this.leg_def.tibia_width / 2, this.leg_def.foot_height / 2 + this.leg_def.foot_height / 2 + this.leg_def.tibia_length + this.leg_def.femur_length + this.torso_def.lower_height + this.torso_def.upper_height + this.head_def.neck_height + this.head_def.head_height / 2);
var head = this.world.CreateBody(this.bd);
- this.fd.shape.SetAsBox(this.head_def.head_width/2, this.head_def.head_height/2);
+ this.fd.shape.SetAsBox(this.head_def.head_width / 2, this.head_def.head_height / 2);
head.CreateFixture(this.fd);
head.SetUserData('head')
// neck joint
var jd = new b2.RevoluteJointDef();
var position = neck.GetPosition().Clone();
- position.y += this.head_def.neck_height/2;
+ position.y += this.head_def.neck_height / 2;
jd.Initialize(head, neck, position);
jd.lowerAngle = -0.1;
jd.upperAngle = 0.2;
@@ -371,17 +390,20 @@ Walker.prototype.createHead = function() {
j.SetUserData('neck_joint')
this.joints.push(j);
- return {head: head, neck: neck};
+ return {
+ head: head,
+ neck: neck
+ };
}
-Walker.prototype.connectParts = function() {
+Walker.prototype.connectParts = function () {
//neck/torso
var jd = new b2.WeldJointDef();
jd.bodyA = this.head.neck;
jd.bodyB = this.torso.upper_torso;
- jd.localAnchorA = new b2.Vec2(0, -this.head_def.neck_height/2);
- jd.localAnchorB = new b2.Vec2(0, this.torso_def.upper_height/2);
+ jd.localAnchorA = new b2.Vec2(0, -this.head_def.neck_height / 2);
+ jd.localAnchorB = new b2.Vec2(0, this.torso_def.upper_height / 2);
jd.referenceAngle = 0;
var j = this.world.CreateJoint(jd);
j.SetUserData('neck-joint')
@@ -390,7 +412,7 @@ Walker.prototype.connectParts = function() {
// torso/arms
var jd = new b2.RevoluteJointDef();
position = this.torso.upper_torso.GetPosition().Clone();
- position.y += this.torso_def.upper_height/2;
+ position.y += this.torso_def.upper_height / 2;
jd.Initialize(this.torso.upper_torso, this.right_arm.upper_arm, position);
jd.lowerAngle = deg2rad(-60);
jd.upperAngle = deg2rad(125);
@@ -417,7 +439,7 @@ Walker.prototype.connectParts = function() {
// torso/legs
var jd = new b2.RevoluteJointDef();
position = this.torso.lower_torso.GetPosition().Clone();
- position.y -= this.torso_def.lower_height/2;
+ position.y -= this.torso_def.lower_height / 2;
jd.Initialize(this.torso.lower_torso, this.right_leg.upper_leg, position);
jd.lowerAngle = deg2rad(-10);
jd.upperAngle = deg2rad(80);
@@ -442,7 +464,7 @@ Walker.prototype.connectParts = function() {
this.joints.push(j);
}
-Walker.prototype.getBodies = function() {
+Walker.prototype.getBodies = function () {
return [
this.head.head,
@@ -464,60 +486,66 @@ Walker.prototype.getBodies = function() {
];
}
-Walker.prototype.randomise = function (n) {
+Walker.prototype.randomise = function (n) {
for (var k = 0; k < this.joints.length; k++) {
- this.joints[k].SetMotorSpeed(Math.randf(-n, n))
+ this.joints[k].SetMotorSpeed(randf(-n, n))
}
}
Walker.prototype.getState = function () {
- var self = this;
+ var self = this;
var state = []
this.bodies
- .forEach((body) => {
+ .forEach((body) => {
// see http://www.box2dflash.org/docs/2.0.2/reference/Box2D/Dynamics/b2Body.html#GetLocalVector()
var t = body.GetTransform() // world transform of the body's origin.
state.push(t.p.x) // world transform of the body's origin.
state.push(t.p.y) // world transform of the body's origin.
state.push(t.q.s) // world transform of the body's origin.
state.push(t.q.c) // world transform of the body's origin.
-
+
var dt = body.GetLinearVelocity() // Get the linear velocity of the center of mass (world).
state.push(dt.x)
state.push(dt.y)
- var lp = self.torso.upper_torso.GetLocalPoint(body.GetWorldCenter())// Get the bodypart position relative to the upper torso
- state.push(lp.x)
+ var lp = self.torso.upper_torso.GetLocalPoint(body.GetWorldCenter()) // Get the bodypart position relative to the upper torso
+ state.push(lp.x)
state.push(lp.y)
state.push(body.GetAngularVelocity()) // the angular velocity in radians/second.
- state.push(body.GetAngle()) // the current world rotation angle in radians.
+ state.push(body.GetAngle()) // the current world rotation angle in radians.
}, [])
- this.joints.forEach(joint => {
+ this.joints.forEach(joint => {
// http://www.box2dflash.org/docs/2.0.2/reference/Box2D/Dynamics/Joints/b2RevoluteJoint.html
state.push(joint.GetJointAngle()) // Get the current joint angle in radians.
state.push(joint.GetJointSpeed()) // Get the current joint angle speed in radians per second
- state.push(joint.GetMotorSpeed())
+ state.push(joint.GetMotorSpeed())
})
return state
}
-Walker.prototype.simulationPreStep = function (motorSpeeds) {
+Walker.prototype.simulationPreStep = function (motorSpeeds) {
// act
for (var k = 0; k < this.joints.length; k++) {
this.joints[k].SetMotorSpeed(motorSpeeds[k] * 10); // action can range from -3 to 3, radians per second
}
- for (let i = 0; i < motorSpeeds.length-this.joints.length; i++) {
+ for (let i = 0; i < motorSpeeds.length - this.joints.length; i++) {
this.grips[i] = motorSpeeds[i] > 0
}
if (motorSpeeds[0] <= 0) this.right_leg.frictionJoint.maxForce = this.right_leg.frictionJoint.maxTorque
if (motorSpeeds[1] <= 0) this.left_leg.frictionJoint.maxForce = this.left_leg.frictionJoint.maxTorque
if (motorSpeeds[2] <= 0) this.left_arm.frictionJoint.maxForce = this.left_arm.frictionJoint.maxTorque
if (motorSpeeds[3] <= 0) this.right_arm.frictionJoint.maxForce = this.right_arm.frictionJoint.maxTorque
- // TODO turn of grups
}
-Walker.prototype.simulationStep = function (motorSpeeds) {
+Walker.prototype.step = function (motorSpeeds) {
+ /*
+ Take one step into the environement
+ @delta (Float) time since the last update
+ @action: (Integer) The action to take (can be null if no action)
+ */
+ this.simulationPreStep(motorSpeeds)
+ this.world.Step(1 / this.config.time_step, this.config.velocity_iterations, this.config.position_iterations);
this.steps++
/* score/reward */
// reward copied from OpenAI Gym Humanoid Walker https://github.com/openai/gym/blob/master/gym/envs/mujoco/humanoid.py
@@ -525,9 +553,9 @@ Walker.prototype.simulationStep = function (motorSpeeds) {
// https://github.com/openai/gym/blob/master/gym/envs/mujoco/assets/humanoidstandup.xml
// reward for keeping head up, compared to feet
- var mean_foot_height = (this.left_leg.foot.GetPosition().y + this.right_leg.foot.GetPosition().y)/2
-
- var head_height_reward = (this.head.head.GetPosition().y - mean_foot_height)* 400; // it's head should be above it's feet 2*(-0.25-2)
+ var mean_foot_height = (this.left_leg.foot.GetPosition().y + this.right_leg.foot.GetPosition().y) / 2
+
+ var head_height_reward = (this.head.head.GetPosition().y - mean_foot_height) * 400; // it's head should be above it's feet 2*(-0.25-2)
// reward for moving one leg beyond the other (stepping)
var left_leg_forward = this.right_leg.foot.GetPosition().x > this.left_leg.foot.GetPosition().x;
@@ -536,7 +564,7 @@ Walker.prototype.simulationStep = function (motorSpeeds) {
// cost for moving joints to unnatural positions (fraction of movement range in the relevant direction)
var jointFractionMovement = j => j.GetJointAngle() > 0 ? j.GetJointAngle() / (j.GetUpperLimit() + 1) : j.GetJointAngle() / (j.GetLowerLimit() - 1)
- var quad_joint_angle_cost = - 0.10 * this.joints.map(j => jointFractionMovement(j) * 1.2)
+ var quad_joint_angle_cost = -0.10 * this.joints.map(j => jointFractionMovement(j) * 1.2)
.reduce((o, v) => o + v * v, 0)
quad_joint_angle_cost = Math.max(quad_joint_angle_cost, -10)
@@ -552,19 +580,10 @@ Walker.prototype.simulationStep = function (motorSpeeds) {
quad_power_cost = Math.max(quad_power_cost, -10)
// Lets be nice, all entities should find overall happiness in what they do
- var bonus_happiness = 5
+ var bonus_happiness = 5
- // we don't have data on external forces, so I will just punish for contact with the ground
- http://blog.sethladd.com/2011/09/box2d-collision-damage-for-javascript.html
- // However we could use a listener or calc force
- // var listener = new b2.ContactListener()
- // listener.PostSolve = function (contact, impulse) {
- // if (contact) console.log(contact)
- // }
- // this.world.SetContactListener(listener)
- // }
var contacts = this.bodies.map(b => b.GetContactList()).filter(b => b).length
- quad_contact_cost = -Math.min(contacts - 4, 10)/2
+ quad_contact_cost = -Math.min(contacts - 4, 10) / 2
this.rewards = {
lin_vel_reward,
@@ -575,15 +594,90 @@ Walker.prototype.simulationStep = function (motorSpeeds) {
head_height_reward,
leg_switch_reward
}
-
- this.reward = Object.values(this.rewards).reduce((tot,v)=>tot+v, 0)/3
+
+ this.reward = Object.values(this.rewards).reduce((tot, v) => tot + v, 0) / 3
var info = {
episodeSteps: this.steps,
- reward:this.reward,
+ reward: this.reward,
position,
...this.rewards
}
var done = 0
+ this.world.ClearForces();
+ console.debug('reward', this.rewards)
return [this.getState(), this.reward, done, info]
}
+
+Walker.prototype.getLastReward = function () {
+ return this.reward
+}
+
+Walker.prototype.render = function (val) {
+ if (val) {
+ this.renderer.setFps(this.config.draw_fps)
+ this.steping = true;
+ } else {
+ this.renderer.setFps(0)
+ this.steping = false;
+ }
+}
+Walker.prototype.reset = function () {
+ /** Reset position to initial or random position TODO */
+ console.log('reset not implemented')
+}
+Walker.prototype.shuffle = function () {
+ /** Reset position to initial or random position TODO */
+ console.log('shuffle not implemented')
+}
+
+
+config = {
+ time_step: 60,
+ simulation_fps: 60,
+ draw_fps: 60,
+ velocity_iterations: 8,
+ position_iterations: 3,
+ max_zoom_factor: 130,
+ min_motor_speed: -2,
+ max_motor_speed: 2,
+ population_size: 1,
+ walker_health: 100,
+ max_floor_tiles: 50,
+ round_length: 1000,
+ min_body_delta: 0,
+ min_leg_delta: 0.0,
+};
+
+var b2 = require('../vendor/jsbox2d')
+var createFloor = require('./floor.js')
+var DDPGAgent = require('./ddpg/ddpg_agent')
+var DDPGAgent = require('./ddpg/ddpg_agent')
+
+
+var world = new b2.World(new b2.Vec2(0, -10))
+floor = createFloor(world, config.max_floor_tiles);
+var env = new Walker(world, floor, config)
+
+var nbActions = env.joints.length + 4
+var stateSize = env.bodies.length * 10 + env.joints.length * 3
+
+var agent = new DDPGAgent(env, {
+ stateSize,
+ nbActions,
+ resetEpisode: true,
+ desiredActionStddev: 0.4,
+ initialStddev: 0.4,
+ actorFirstLayerSize: 128,
+ actorSecondLayerSize: 64,
+ criticFirstLayerSSize: 128,
+ criticFirstLayerASize: 128,
+ criticSecondLayerSize: 64,
+ nbEpochs: 1000
+});
+agent.train(true);
+
+// setInterval(() => {
+// walker.step()
+// console.log(walker.getState())
+// }, 100)
diff --git a/package.json b/package.json
index 1e5a6e4..997af8d 100644
--- a/package.json
+++ b/package.json
@@ -3,7 +3,12 @@
"version": "0.0.1",
"description": "TODO: Write a project description",
"main": "index.js",
- "dependencies": {},
+ "dependencies": {
+ "@tensorflow/tfjs": "^0.14.0",
+ "canvas": "^2.1.0",
+ "jsdom": "^13.0.0",
+ "phaser": "^3.15.1"
+ },
"devDependencies": {},
"scripts": {
"test": "echo \"Error: no test specified\" && exit 1"