diff --git a/README.md b/README.md index 960f337..479e083 100644 --- a/README.md +++ b/README.md @@ -37,4 +37,4 @@ TODO: Write usage instructions # Credits - The walker code is adapted from http://rednuht.org/genetic_walkers/ -- The refinforcement learning code uses neurojs +- DDPG code from metacar diff --git a/js/ddpg/ddpg.js b/js/ddpg/ddpg.js new file mode 100644 index 0000000..4c0eafd --- /dev/null +++ b/js/ddpg/ddpg.js @@ -0,0 +1,275 @@ +const { tf } = require('./tf_import') +const {copyModel, Actor, Critic, assignAndStd } = require('./models') + +function logTfMemory(){ + let mem = tf.memory(); + console.log("numBytes:" + mem.numBytes + + "\nnumBytesInGPU:" + mem.numBytesInGPU + + "\nnumDataBuffers:" + mem.numDataBuffers + + "\nnumTensors:" + mem.numTensors); +} + +// This class is called from js/DDPG/ddpg_agent.js +class DDPG { + + /** + * @param config (Object) + * @param actor (Actor class) + * @param critic (Critic class) + * @param memory (Memory class) + * @param noise (Noise class) + */ + constructor(actor, critic, memory, noise, config){ + this.actor = actor; + this.critic = critic; + this.memory = memory; + this.noise = noise; + this.config = config; + this.tfGamma = tf.scalar(config.gamma); + + // Inputs + this.obsInput = tf.input({batchShape: [null, this.config.stateSize]}); + this.actionInput = tf.input({batchShape: [null, this.config.nbActions]}); + + // Randomly Initialize actor network μ(s) + this.actor.buildModel(this.obsInput); + // Randomly Initialize critic network Q(s, a) + this.critic.buildModel(this.obsInput, this.actionInput); + + + // Define in js/DDPG/models.js + // Init target network Q' and μ' with the same weights + this.actorTarget = copyModel(this.actor, Actor); + this.criticTarget = copyModel(this.critic, Critic); + // Perturbed Actor (See parameter space noise Exploration paper) + this.perturbedActor = copyModel(this.actor, Actor); + //this.adaptivePerturbedActor = copyModel(this.actor, Actor); + + this.setLearningOp(); + } + + setLearningOp(){ + this.criticWithActor = (tfState) => { + return tf.tidy(() => { + const tfAct = this.actor.predict(tfState); + return this.critic.predict(tfState, tfAct); + }); + }; + this.criticTargetWithActorTarget = (tfState) => { + return tf.tidy(() => { + const tfAct = this.actorTarget.predict(tfState); + return this.criticTarget.predict(tfState, tfAct); + }); + }; + + this.actorOptimiser = tf.train.adam(this.config.actorLr); + this.criticOptimiser = tf.train.adam(this.config.criticLr); + + this.criticWeights = []; + for (let w = 0; w < this.critic.model.trainableWeights.length; w++){ + this.criticWeights.push(this.critic.model.trainableWeights[w].val); + } + this.actorWeights = []; + for (let w = 0; w < this.actor.model.trainableWeights.length; w++){ + this.actorWeights.push(this.actor.model.trainableWeights[w].val); + } + + assignAndStd(this.actor, this.perturbedActor, this.noise.currentStddev, this.config.seed); + } + + + /** + * Distance Measure for DDPG + * See parameter space noise Exploration paper + * @param observations (Tensor2d) Observations + */ + distanceMeasure(observations) { + return tf.tidy(() => { + const pertubedPredictions = this.perturbedActor.model.predict(observations); + const predictions = this.actor.model.predict(observations); + + const distance = tf.square(pertubedPredictions.sub(predictions)).mean().sqrt(); + return distance; + }); + } + + /** + * AdaptParamNoise + */ + adaptParamNoise(){ + const batch = this.memory.getBatch(this.config.batchSize); + if (batch.obs0.length == 0){ + assignAndStd(this.actor, this.perturbedActor, this.noise.currentStddev, this.config.seed); + return [0]; + } + + let distanceV = null; + + if (batch.obs0.length > 0){ + const tfObs0 = tf.tensor2d(batch.obs0); + const distance = this.distanceMeasure(tfObs0); + + assignAndStd(this.actor, this.perturbedActor, this.noise.currentStddev, this.config.seed); + + distanceV = distance.buffer().values; + this.noise.adapt(distanceV[0]); + + distance.dispose(); + tfObs0.dispose(); + } + + return distanceV; + } + + /** + * Get the estimation of the Q value given the state + * and the action + * @param state number[] + * @param action [a, steering] + */ + getQvalue(state, a){ + const st = tf.tensor2d([state]); + const tfa = tf.tensor2d([a]); + const q = this.critic.model.predict([st, tfa]); + const v = q.buffer().values + st.dispose(); + tfa.dispose(); + q.dispose(); + return v[0]; + } + + /** + * @param observation (tf.tensor2d) + * @return (tf.tensor1d) + */ + predict(observation){ + const tfActions = this.actor.model.predict(observation); + return tfActions; + } + + /** + * @param observation (tf.tensor2d) + * @return (tf.tensor1d) + */ + perturbedPrediction(observation){ + const tfActions = this.perturbedActor.model.predict(observation); + return tfActions; + } + + /** + * Update the two target network + */ + targetUpdate(){ + // Define in js/DDPG/models.js + targetUpdate(this.criticTarget, this.critic, this.config); + targetUpdate(this.actorTarget, this.actor, this.config); + } + + trainCritic(batch, tfActions, tfObs0, tfObs1, tfRewards, tfTerminals){ + + let costs; + + const criticLoss = this.criticOptimiser.minimize(() => { + const tfQPredictions0 = this.critic.model.predict([tfObs0, tfActions]); + const tfQPredictions1 = this.criticTargetWithActorTarget(tfObs1); + + const tfQTargets = tfRewards.add(tf.scalar(1).sub(tfTerminals).mul(this.tfGamma).mul(tfQPredictions1)); + + const erros = tf.sub(tfQTargets, tfQPredictions0).square(); + costs = erros.buffer().values; + + return erros.mean(); + }, true, this.criticWeights); + + // For experience Replay + this.memory.appendBackWithCost(batch, costs); + + const loss = criticLoss.buffer().values[0]; + criticLoss.dispose(); + + targetUpdate(this.criticTarget, this.critic, this.config); + + return loss; + } + + trainActor(tfObs0){ + + const actorLoss = this.actorOptimiser.minimize(() => { + const tfQPredictions0 = this.criticWithActor(tfObs0); + return tf.mean(tfQPredictions0).mul(tf.scalar(-1.)) + }, true, this.actorWeights); + + targetUpdate(this.actorTarget, this.actor, this.config); + + const loss = actorLoss.buffer().values[0]; + actorLoss.dispose(); + + return loss; + } + + getTfBatch(){ + // Get batch + const batch = this.memory.popBatch(this.config.batchSize); + // Convert to tensors + const tfActions = tf.tensor2d(batch.actions); + const tfObs0 = tf.tensor2d(batch.obs0); + const tfObs1 = tf.tensor2d(batch.obs1); + const _tfRewards = tf.tensor1d(batch.rewards); + const _tfTerminals = tf.tensor1d(batch.terminals); + + const tfRewards = _tfRewards.expandDims(1); + const tfTerminals = _tfTerminals.expandDims(1); + + _tfRewards.dispose(); + _tfTerminals.dispose(); + + return { + batch, tfActions, tfObs0, tfObs1, tfRewards, tfTerminals + } + } + + optimizeCritic(){ + const {batch, tfActions, tfObs0, tfObs1, tfRewards, tfTerminals} = this.getTfBatch(); + + const loss = this.trainCritic(tfActions, tfObs0, tfObs1, tfRewards, tfTerminals); + + tfActions.dispose(); + tfObs0.dispose(); + tfObs1.dispose(); + tfRewards.dispose(); + tfTerminals.dispose(); + + return loss; + } + + optimizeActor(){ + const {batch, tfActions, tfObs0, tfObs1, tfRewards, tfTerminals} = this.getTfBatch(); + + const loss = this.trainActor(tfObs0); + + tfActions.dispose(); + tfObs0.dispose(); + tfObs1.dispose(); + tfRewards.dispose(); + tfTerminals.dispose(); + + return loss; + } + + optimizeCriticActor(){ + const {batch, tfActions, tfObs0, tfObs1, tfRewards, tfTerminals} = this.getTfBatch(); + + const lossC = this.trainCritic(batch, tfActions, tfObs0, tfObs1, tfRewards, tfTerminals); + const lossA = this.trainActor(tfObs0); + + tfActions.dispose(); + tfObs0.dispose(); + tfObs1.dispose(); + tfRewards.dispose(); + tfTerminals.dispose(); + + return {lossC, lossA}; + } +} + +module.exports = { logTfMemory, DDPG } diff --git a/js/ddpg/ddpg_agent.js b/js/ddpg/ddpg_agent.js new file mode 100644 index 0000000..bda770b --- /dev/null +++ b/js/ddpg/ddpg_agent.js @@ -0,0 +1,241 @@ +const { tf } = require('./tf_import') +const AdaptiveParamNoiseSpec = require('./noise') +const PrioritizedMemory = require('./prioritized_memory') +const { Actor, Critic, } = require('./models') +const { DDPG, logTfMemory } = require('./ddpg') + +// This class is called from js/DDPG/index.js +class DDPGAgent { + + /** + * @param env (metacar.env) Set in js/DDPG/index.js + */ + constructor(env, config){ + + this.stopTraining = false; + this.env = env; + + config = config || {}; + // Default Config + this.config = { + "stateSize": config.stateSize || 17, + "nbActions": config.nbActions || 2, + "seed": config.seed || 0, + "batchSize": config.batchSize || 128, + "actorLr": config.actorLr || 0.0001, + "criticLr": config .criticLr || 0.001, + "memorySize": config.memorySize || 30000, + "gamma": config.gamme || 0.99, + "noiseDecay": config.noiseDecay || 0.99, + "rewardScale": config.rewardScale || 1, + "nbEpochs": config.nbEpochs || 200, + "nbEpochsCycle": config.nbEpochsCycle || 10, + "nbTrainSteps": config.nbTrainSteps || 110, + "tau": config.tau || 0.008, + "initialStddev": config.initialStddev || 0.1, + "desiredActionStddev": config.desiredActionStddev || 0.1, + "adoptionCoefficient": config.adoptionCoefficient || 1.01, + "actorFirstLayerSize": config.actorFirstLayerSize || 64, + "actorSecondLayerSize": config.actorSecondLayerSize || 32, + "criticFirstLayerSSize": config.criticFirstLayerSSize || 64, + "criticFirstLayerASize": config.criticFirstLayerASize || 64, + "criticSecondLayerSize": config.criticSecondLayerSize || 32, + "maxStep": config.maxStep || 800, + "stopOnRewardError": config.stopOnRewardError != undefined ? config.stopOnRewardError:true, + "resetEpisode": config.resetEpisode != undefined ? config.resetEpisode:false, + "saveDuringTraining": config.saveDuringTraining || false, + "saveInterval": config.saveInterval || 20 + }; + this.epoch = 0; + // From js/DDPG/noise.js + this.noise = new AdaptiveParamNoiseSpec(this.config); + + // Configure components. + + // Buffer replay + // The baseline use 1e6 but this size should be enough for this problem + this.memory = new PrioritizedMemory(this.config.memorySize); + // Actor and Critic are from js/DDPG/models.js + this.actor = new Actor(this.config); + this.critic = new Critic(this.config); + + // Seed javascript + // Math.seedrandom(0); + + this.rewardsList = []; + this.epiDuration = []; + + // DDPG + this.ddpg = new DDPG(this.actor, this.critic, this.memory, this.noise, this.config); + } + + save(name){ + /* + Save the network + */ + this.ddpg.critic.model.save('downloads://critic-' + name); + this.ddpg.actor.model.save('downloads://actor-'+ name); + } + + async restore(folder, name){ + /* + Restore the weights of the network + */ + const critic = await tf.loadModel('https://metacar-project.com/public/models/'+folder+'/critic-'+name+'.json'); + const actor = await tf.loadModel("https://metacar-project.com/public/models/"+folder+"/actor-"+name+".json"); + + this.ddpg.critic = copyFromSave(critic, Critic, this.config, this.ddpg.obsInput, this.ddpg.actionInput); + this.ddpg.actor = copyFromSave(actor, Actor, this.config, this.ddpg.obsInput, this.ddpg.actionInput); + + // Define in js/DDPG/models.js + // Init target network Q' and μ' with the same weights + this.ddpg.actorTarget = copyModel(this.ddpg.actor, Actor); + this.ddpg.criticTarget = copyModel(this.ddpg.critic, Critic); + // Perturbed Actor (See parameter space noise Exploration paper) + this.ddpg.perturbedActor = copyModel(this.ddpg.actor, Actor); + //this.adaptivePerturbedActor = copyModel(this.actor, Actor); + this.ddpg.setLearningOp(); + } + + /** + * Play one step + */ + play(){ + // Get the current state + const state = this.env.getState(); + // Pick an action + const tfActions = this.ddpg.predict(tf.tensor2d([state])); + const actions = tfActions.buffer().values; + agent.env.step(actions); + tfActions.dispose(); + } + + /** + * Get the estimation of the Q value given the state + * and the action + * @param state number[] + * @param action [a, steering] + */ + getQvalue(state, a){ + return this.ddpg.getQvalue(state, a); + } + + stop(){ + this.stopTraining = true; + } + + /** + * Step into the training environement + * @param tfPreviousStep (tf.tensor2d) Current state + * @param mPreviousStep number[] + * @return {done, state} One boolean and the new state + */ + stepTrain(tfPreviousStep, mPreviousStep){ + // Get actions + const tfActions = this.ddpg.perturbedPrediction(tfPreviousStep); + // Step in the environment with theses actions + let mAcions = tfActions.buffer().values; + let mReward = this.env.step(mAcions); + this.rewardsList.push(mReward); + // Get the new observations + let mState = this.env.getState(); + let tfState = tf.tensor2d([mState]); + let mDone = 0; + if (mReward == -1 && this.config.stopOnRewardError){ + mDone = 1; + } + // Add the new tuple to the buffer + this.ddpg.memory.append(mPreviousStep, mAcions, mReward, mState, mDone); + // Dispose tensors + tfPreviousStep.dispose(); + tfActions.dispose(); + + return {mDone, mState, tfState}; + } + + /** + * Optimize models and log states + */ + _optimize(){ + this.ddpg.noise.desiredActionStddev = Math.max(0.1, this.config.noiseDecay * this.ddpg.noise.desiredActionStddev); + let lossValuesCritic = []; + let lossValuesActor = []; + console.time("Training"); + for (let t=0; t < this.config.nbTrainSteps; t++){ + let {lossC, lossA} = this.ddpg.optimizeCriticActor(); + lossValuesCritic.push(lossC); + lossValuesActor.push(lossA); + } + console.timeEnd("Training"); + console.log("desiredActionStddev:", this.ddpg.noise.desiredActionStddev); + setMetric("CriticLoss", mean(lossValuesCritic)); + setMetric("ActorLoss", mean(lossValuesActor)); + } + + /** + * Train DDPG Agent + */ + async train(realTime){ + this.stopTraining = false; + // One epoch + for (this.epoch; this.epoch < this.config.nbEpochs; this.epoch++){ + // Perform cycles. + this.rewardsList = []; + this.stepList = []; + this.distanceList = []; + // document.getElementById("trainingProgress").innerHTML = "Progression: "+this.epoch+"/"+this.config.nbEpochs+"
"; + console.log("Progression: "+this.epoch+"/"+this.config.nbEpochs+" epochs") + for (let c=0; c < this.config.nbEpochsCycle; c++){ + if (c%10==0){ logTfMemory(); } + let mPreviousStep = this.env.getState(); + let tfPreviousStep = tf.tensor2d([mPreviousStep]); + let step = 0; + + console.time("LoopTime"); + for (step=0; step < this.config.maxStep; step++){ + let rel = this.stepTrain(tfPreviousStep, mPreviousStep); + mPreviousStep = rel.mState; + tfPreviousStep = rel.tfState; + if (rel.mDone && this.config.stopOnRewardError){ + break; + } + if (this.stopTraining){ + this.env.render(true); + return; + } + if (realTime && step % 10 == 0) + await tf.nextFrame(); + } + this.stepList.push(step); + console.timeEnd("LoopTime"); + let distance = this.ddpg.adaptParamNoise(); + this.distanceList.push(distance[0]); + + if (this.config.resetEpisode){ + this.env.reset(); + } + this.env.shuffle({cars: false}); + tfPreviousStep.dispose(); + console.log("e="+ this.epoch +", c="+c); + + await tf.nextFrame(); + } + if (this.epoch > 5){ + this._optimize(); + } + if (this.config.saveDuringTraining && this.epoch % this.config.saveInterval == 0 && this.epoch != 0){ + this.save("model-ddpg-traffic-epoch-"+this.epoch); + } + setMetric("Reward", mean(this.rewardsList)); + setMetric("EpisodeDuration", mean(this.stepList)); + setMetric("NoiseDistance", mean(this.distanceList)); + await tf.nextFrame(); + } + + + this.env.render(true); + } + +}; + +module.exports = DDPGAgent diff --git a/js/ddpg/models.js b/js/ddpg/models.js new file mode 100644 index 0000000..407f1dd --- /dev/null +++ b/js/ddpg/models.js @@ -0,0 +1,249 @@ +const { tf } = require('./tf_import') + +/** + * Copy a model + * @param model Actor|Critic instance + * @param instance Actor|Critic + * @return Copy of the model + */ +function copyFromSave(model, instance, config, obs, action){ + return tf.tidy(() => { + nModel = new instance(config); + // action might be not required + nModel.buildModel(obs, action); + const weights = model.weights; + for (let m=0; m < weights.length; m++){ + nModel.model.weights[m].val.assign(weights[m].val); + } + return nModel; + }) +} + + +/** + * Copy a model + * @param model Actor|Critic instance + * @param instance Actor|Critic + * @return Copy of the model + */ +function copyModel(model, instance){ + return tf.tidy(() => { + nModel = new instance(model.config); + // action might be not required + nModel.buildModel(model.obs, model.action); + const weights = model.model.weights; + for (let m=0; m < weights.length; m++){ + nModel.model.weights[m].val.assign(weights[m].val); + } + return nModel; + }) +} + +/** + * Copy the value of of the model into the perturbedModel and + * add a random pertubation + * @param model Actor|Critic instance + * @param perturbedActor Actor|Critic instance + * @param stddev (number) + * @return Copy of the model + */ +function assignAndStd(model, perturbedModel, stddev, seed){ + return tf.tidy(() => { + const weights = model.model.trainableWeights; + for (let m=0; m < weights.length; m++){ + let shape = perturbedModel.model.trainableWeights[m].val.shape; + let randomTensor = tf.randomNormal(shape, 0, stddev, "float32", seed); + let nValue = weights[m].val.add(randomTensor); + perturbedModel.model.trainableWeights[m].val.assign(nValue); + } + }); +} + +/** + * Update the target models + * @param target Actor|Critic instance + * @param perturbedActor Actor|Critic instance + * @param config (Object) + * @return Copy of the model + */ +function targetUpdate(target, original, config){ + return tf.tidy(() => { + const originalW = original.model.trainableWeights; + const targetW = target.model.trainableWeights; + + const one = tf.scalar(1); + const tau = tf.scalar(config.tau); + + for (let m=0; m < originalW.length; m++){ + const lastValue = target.model.trainableWeights[m].val.clone(); + let nValue = tau.mul(originalW[m].val).add(targetW[m].val.mul(one.sub(tau))); + target.model.trainableWeights[m].val.assign(nValue); + const diff = lastValue.sub(target.model.trainableWeights[m].val).mean().buffer().values; + if (diff[0] == 0){ + console.warn("targetUpdate: Nothing have been changed!") + } + } + }); +} + + +class Actor{ + + /** + @param config (Object) + */ + constructor(config) { + this.stateSize = config.stateSize; + this.nbActions = config.nbActions; + this.layerNorm = config.layerNorm; + + this.firstLayerSize = config.actorFirstLayerSize; + this.secondLayerSize = config.actorSecondLayerSize; + + this.seed = config.seed; + this.config = config; + this.obs = null; + } + + /** + * + * @param obs tf.input + */ + buildModel(obs){ + this.obs = obs; + + // First layer + this.firstLayer = tf.layers.dense({ + units: this.firstLayerSize, + kernelInitializer: tf.initializers.glorotUniform({seed: this.seed}), + activation: 'relu', + useBias: true, + biasInitializer: "zeros" + }); + // Second layer + this.secondLayer = tf.layers.dense({ + units: this.secondLayerSize, + kernelInitializer: tf.initializers.glorotUniform({seed: this.seed}), + activation: 'relu', + useBias: true, + biasInitializer: "zeros" + }); + // Ouput layer + this.outputLayer = tf.layers.dense({ + units: this.nbActions, + kernelInitializer: tf.initializers.randomUniform({ + minval: 0.003, maxval: 0.003, seed: this.seed}), + activation: 'tanh', + useBias: true, + biasInitializer: "zeros" + }); + // Actor prediction + this.predict = (tfState) => { + return tf.tidy(() => { + if (tfState){ + obs = tfState; + } + + let l1 = this.firstLayer.apply(obs); + let l2 = this.secondLayer.apply(l1); + + return this.outputLayer.apply(l2); + }); + } + const output = this.predict(); + this.model = tf.model({inputs: obs, outputs: output}); + } +}; + +class Critic { + + /** + * @param config (Object) + */ + constructor(config) { + this.stateSize = config.stateSize; + this.nbActions = config.nbActions; + this.layerNorm = config.layerNorm; + + this.firstLayerSSize = config.criticFirstLayerSSize + this.firstLayerASize = config.criticFirstLayerASize; + this.secondLayerSize = config.criticSecondLayerSize; + + this.seed = config.seed; + this.config = config; + this.obs = null; + this.action = null; + } + + /** + * + * @param obs tf.input + * @param action tf.input + */ + buildModel(obs, action){ + this.obs = obs; + this.action = action; + + // Used to merged the two first Layer later. + this.add = tf.layers.add(); + + // First layer + this.firstLayerS = tf.layers.dense({ + units: this.firstLayerSSize, + kernelInitializer: tf.initializers.glorotUniform({seed: this.seed}), + activation: 'linear', // relu is add later + useBias: true, + biasInitializer: "zeros" + }); + // First layer + this.firstLayerA = tf.layers.dense({ + units: this.firstLayerASize, + kernelInitializer: tf.initializers.glorotUniform({seed: this.seed}), + activation: 'linear', // relu is add later + useBias: true, + biasInitializer: "zeros" + }); + // Second layer + this.secondLayer = tf.layers.dense({ + units: this.secondLayerSize, + kernelInitializer: tf.initializers.glorotUniform({seed: this.seed}), + activation: 'relu', + useBias: true, + biasInitializer: "zeros" + }); + + // Ouput layer + this.outputLayer = tf.layers.dense({ + units: 1, + kernelInitializer: tf.initializers.randomUniform({ + minval: 0.003, maxval: 0.003, seed: this.seed}), + activation: 'linear', + useBias: true, + biasInitializer: "zeros" + }); + + // Critic prediction + this.predict = (tfState, tfActions) => { + return tf.tidy(() => { + if (tfState && tfActions){ + obs = tfState; + action = tfActions; + } + + let l1A = this.firstLayerA.apply(action); + let l1S = this.firstLayerS.apply(obs) + // Merged layers + let concat = this.add.apply([l1A, l1S]) + + let l2 = this.secondLayer.apply(concat); + + return this.outputLayer.apply(l2); + }); + } + + const output = this.predict(); + this.model = tf.model({inputs: [obs, action], outputs: output}); + } +}; + +module.exports = {Actor, Critic, copyFromSave, copyModel, assignAndStd, targetUpdate} diff --git a/js/ddpg/noise.js b/js/ddpg/noise.js new file mode 100644 index 0000000..2b19ab4 --- /dev/null +++ b/js/ddpg/noise.js @@ -0,0 +1,43 @@ +/** + * Noise class + * The original baseline is made of three noise + * AdaptiveParamNoiseSpec, ActionNoise and NormalActionNoise + * Only AdaptiveParamNoiseSpec is implemented for now + * See "C Adapative Scaling" Page 14 in the paper. + */ + +class AdaptiveParamNoiseSpec { + + /** + * @param conf Object + * conf.initialStddev: 0.1 default // σ + * conf.desiredActionStddev: 0.1 default // δ + * conf.adoptionCoefficient: 1.01 default // α + */ + constructor(conf){ + conf = conf || {}; + this.initialStddev = conf.initialStddev || 0.4; + this.desiredActionStddev = conf.desiredActionStddev || 0.4; + this.adoptionCoefficient = conf.adoptionCoefficient || 1.01; + this.currentStddev = this.initialStddev; + } + + /** + * The distance from the Adaptive scaling + * @param distance number + */ + adapt(distance){ + // if d(π, _π_) > δ then σ = σ/α + if (distance > this.desiredActionStddev){ + // Decrease σ + this.currentStddev /= this.adoptionCoefficient; + } + else{ + // σ = σ*α + // Increase σ + this.currentStddev *= this.adoptionCoefficient; + } + } +}; + +module.exports = AdaptiveParamNoiseSpec diff --git a/js/ddpg/prioritized_memory.js b/js/ddpg/prioritized_memory.js new file mode 100644 index 0000000..0979a76 --- /dev/null +++ b/js/ddpg/prioritized_memory.js @@ -0,0 +1,242 @@ + +class PrioritizedMemory { + + /** + * @param maxlen (number) Buffer limit + */ + constructor(maxlen){ + this.maxlen = maxlen; + this.buffer = []; + this.priorBuffer = []; + } + + /** + * Sample a batch + * @param batchSize (number) + * @return batch [] + */ + getBatch(batchSize){ + const batch = { + 'obs0': [], + 'obs1': [], + 'rewards': [], + 'actions': [], + 'terminals': [], + }; + + if (batchSize > this.priorBuffer.length){ + console.warn("The size of the replay buffer is < to the batchSize. Return empty batch."); + return batch; + } + + for (let b=0; b < batchSize/2; b++){ + let id = Math.floor(Math.random() * this.priorBuffer.length); + batch.obs0.push(this.priorBuffer[id].obs0); + batch.obs1.push(this.priorBuffer[id].obs1); + batch.rewards.push(this.priorBuffer[id].reward); + batch.actions.push(this.priorBuffer[id].action); + batch.terminals.push(this.priorBuffer[id].terminal); + } + return batch + } + + _bufferBatch(batchSize){ + const batch = { + 'obs0': [], + 'obs1': [], + 'rewards': [], + 'actions': [], + 'terminals': [], + }; + + for (let b=0; b < batchSize/2; b++){ + let nElem = this.buffer.pop(); + batch.obs0.push(nElem.obs0); + batch.obs1.push(nElem.obs1); + batch.rewards.push(nElem.reward); + batch.actions.push(nElem.action); + batch.terminals.push(nElem.terminal); + } + + for (let b=0; b < batchSize/2; b++){ + let id = Math.floor(Math.random() * this.buffer.length); + batch.obs0.push(this.buffer[id].obs0); + batch.obs1.push(this.buffer[id].obs1); + batch.rewards.push(this.buffer[id].reward); + batch.actions.push(this.buffer[id].action); + batch.terminals.push(this.buffer[id].terminal); + this.buffer.splice(id, 1); + } + + return batch + } + + _addRandomBufferBatch(batchSize, batch){ + for (let b=0; b < batchSize; b++){ + let id = Math.floor(Math.random() * this.buffer.length); + batch.obs0.push(this.buffer[id].obs0); + batch.obs1.push(this.buffer[id].obs1); + batch.rewards.push(this.buffer[id].reward); + batch.actions.push(this.buffer[id].action); + batch.terminals.push(this.buffer[id].terminal); + this.buffer.splice(id, 1); + } + return batch + } + + /** + * Sample a batch + * @param batchSize (number) + * @return batch [] + */ + popBatch(batchSize){ + let originalBatchSize = batchSize; + let priorBufferBatchSize; + let bufferBatchSize; + if (batchSize % 2 != 0){ + console.warn("Batch size should be a even.") + } + if (this.priorBuffer.length < batchSize/2){ + //console.log("get full batch from buffer"); + const batch = this._bufferBatch(batchSize); + console.assert(batch.obs0.length == batchSize); + return batch; + } + const batch = { + 'obs0': [], + 'obs1': [], + 'rewards': [], + 'actions': [], + 'terminals': [], + }; + if (batchSize > this.length){ + console.warn("The size of the replay buffer is < to the batchSize. Return empty batch."); + return batch; + } + + if (this.buffer.length > 0){ + //console.log("Get half of prior and other from buffer."); + batchSize = batchSize / 2; + } + else{ + //console.log("Get all from priorBuffer"); + } + + for (let b=0; b < batchSize; b++){ + let id = Math.floor(Math.random() * this.priorBuffer.length); + batch.obs0.push(this.priorBuffer[id].obs0); + batch.obs1.push(this.priorBuffer[id].obs1); + batch.rewards.push(this.priorBuffer[id].reward); + batch.actions.push(this.priorBuffer[id].action); + batch.terminals.push(this.priorBuffer[id].terminal); + this.priorBuffer.splice(id, 1); + } + + if (this.buffer.length > 0){ + this._addRandomBufferBatch(batchSize, batch); + } + console.assert(batch.obs0.length == originalBatchSize); + return batch + } + + _insert(element, array) { + if (array.length == 0 || element.cost < array[0].cost || array[0].cost == null){ + array.unshift(element); + return array; + } + array.splice(this._locationOf(element, array) + 1, 0, element); + return array; + } + + _locationOf(element, array, start, end) { + start = start || 0; + end = end || array.length; + + var pivot = parseInt(start + (end - start) / 2, 10); + + if (end-start <= 1 || array[pivot] === element) return pivot; + + if (array[pivot].cost != null && array[pivot].cost < element.cost) { + return this._locationOf(element, array, pivot, end); + } else { + return this._locationOf(element, array, start, pivot); + } + } + + /** + * @param batch (Object) from getBatch() + * @param cost (number) Cost associated with each row of the batch + */ + appendBackWithCost(batch, costs){ + for (let b=0; b < batch.obs0.length; b++){ + if (this.buffer.length == this.maxlen){ + this.buffer.shift(); + } + this._insert({ + obs0: batch.obs0[b], + action: batch.actions[b], + reward: batch.rewards[b], + obs1: batch.obs1[b], + terminal: batch.terminals[b], + cost: costs[b] + }, this.buffer); + } + console.assert(this.buffer.length <= this.maxlen); + } + + /** + * @param obs0 [] + * @param action (number) + * @param reward (number) + * @param obs1 [] + * @param terminal1 (boolean) + */ + append(obs0, action, reward, obs1, terminal){ + if (this.priorBuffer.length == this.maxlen){ + this.priorBuffer.shift(); + } + this.priorBuffer.push({ + obs0: obs0, + action: action, + reward: reward, + obs1: obs1, + terminal: terminal, + cost: null + }); + console.assert(this.priorBuffer.length <= this.maxlen); + } +} +/* +var mem = new Memory(20000); +Math.seedrandom(0); +console.assert(mem.length == 0); + +var array = []; +for (let i=1; i < 40000; i++){ + mem.append("obs0-"+i, "action-"+i, "reward-"+i, "obs1-"+i, "terminal-"+i); +} + +console.assert(mem.length == 20000); +console.assert(mem.list[0].obs0 == "obs0-20000"); +console.assert(mem.list[19999].obs0 == "obs0-39999"); + +let batch = mem.getBatch(32); + +console.assert(batch.obs0.length == 32); +console.assert(mem.length == 20000 - 32); + +let costs = []; +for (i=31; i >= 0; i--){ + costs.push(i); +} +mem.appendBackWithCost(batch, costs); + +console.log(mem.list); + +/* +for (let i=1; i < 64; i++){ + mem.append("obs0-"+i, "action-"+i, "reward-"+i, "obs1-"+i, "terminal-"+i); +} +*/ + +module.exports = PrioritizedMemory diff --git a/js/ddpg/tf_import.js b/js/ddpg/tf_import.js new file mode 100644 index 0000000..862ebeb --- /dev/null +++ b/js/ddpg/tf_import.js @@ -0,0 +1,2 @@ +const tf = require('@tensorflow/tfjs') +module.exports = { tf } diff --git a/js/draw.js b/js/draw.js index 9a3a895..b8607e3 100644 --- a/js/draw.js +++ b/js/draw.js @@ -1,147 +1,154 @@ -drawInit = function() { - globals.main_screen = document.getElementById("main_screen"); - globals.ctx = main_screen.getContext("2d"); - resetCamera(); -} +class Renderer { + constructor(config, walker, floor) { + this.config = config + this.walkers = [walker] + this.floor = foor -resetCamera = function() { - globals.zoom = config.max_zoom_factor; - globals.translate_x = 0; - globals.translate_y = 280; -} - -setFps = function(fps) { - config.draw_fps = fps; - if(globals.draw_interval) - clearInterval(globals.draw_interval); - if(fps > 0 && config.simulation_fps > 0) { - globals.draw_interval = setInterval(drawFrame, Math.round(1000/config.draw_fps)); + this.main_screen = document.getElementById("main_screen"); + this.ctx = main_screen.getContext("2d"); + resetCamera(); } -} -drawFrame = function() { - var minmax = getMinMaxDistance(); - globals.target_zoom = Math.min(config.max_zoom_factor, getZoom(minmax.min_x, minmax.max_x + 4, minmax.min_y + 2, minmax.max_y + 2.5)); - globals.zoom += 0.1*(globals.target_zoom - globals.zoom); - globals.translate_x += 0.1*(1.5-minmax.min_x - globals.translate_x); - globals.translate_y += 0.3*(minmax.min_y*globals.zoom + 280 - globals.translate_y); - //globals.translate_y = minmax.max_y*globals.zoom + 150; - globals.ctx.clearRect(0, 0, globals.main_screen.width, globals.main_screen.height); - globals.ctx.save(); - globals.ctx.translate(globals.translate_x*globals.zoom, globals.translate_y); - globals.ctx.scale(globals.zoom, -globals.zoom); - drawFloor(); - for(var k = config.population_size - 1; k >= 0 ; k--) { - drawWalker(globals.walkers[k]); + resetCamera() { + this.zoom = config.max_zoom_factor; + this.translate_x = 0; + this.translate_y = 280; } - globals.ctx.restore(); -} -drawFloor = function() { - globals.ctx.strokeStyle = "#444"; - globals.ctx.lineWidth = 1/globals.zoom; - globals.ctx.beginPath(); - var floor_fixture = globals.floor.GetFixtureList(); - globals.ctx.moveTo(floor_fixture.m_shape.m_vertices[0].x, floor_fixture.m_shape.m_vertices[0].y); - for(var k = 1; k < floor_fixture.m_shape.m_vertices.length; k++) { - globals.ctx.lineTo(floor_fixture.m_shape.m_vertices[k].x, floor_fixture.m_shape.m_vertices[k].y); - } - globals.ctx.stroke(); -} - -drawWalker = function (walker) { - var hue = walker.hue || 240 - globals.ctx.strokeStyle = "hsl(" + hue + ",100%,0%)"; - globals.ctx.fillStyle = "hsl("+hue+",45%,"+(100-15*walker.health/config.walker_health)+"%)"; - globals.ctx.lineWidth = 1/globals.zoom; - - // left legs and arms first - drawRect(walker.left_leg.lower_leg); - drawRect(walker.left_leg.upper_leg); - drawRect(walker.left_arm.upper_arm); - drawRect(walker.left_arm.lower_arm); - - globals.ctx.lineWidth = walker.left_leg.frictionJoint.maxForce? 4/globals.zoom : 1/globals.zoom; - drawRect(walker.left_leg.foot); - globals.ctx.lineWidth = 1/globals.zoom; - - globals.ctx.lineWidth = walker.left_arm.frictionJoint.maxForce? 4/globals.zoom : 1/globals.zoom; - drawRect(walker.left_arm.hand); - globals.ctx.lineWidth = 1/globals.zoom; - - // head - drawRect(walker.head.neck); - drawRect(walker.head.head); - - // torso - drawRect(walker.torso.lower_torso); - drawRect(walker.torso.upper_torso); - - // right legs and arms - drawRect(walker.right_leg.upper_leg); - drawRect(walker.right_leg.lower_leg); - drawRect(walker.right_arm.upper_arm); - drawRect(walker.right_arm.lower_arm); - - globals.ctx.lineWidth = walker.right_leg.frictionJoint.maxForce? 4/globals.zoom : 1/globals.zoom; - drawRect(walker.right_leg.foot); - globals.ctx.lineWidth = 1/globals.zoom; - globals.ctx.lineWidth = walker.right_arm.frictionJoint.maxForce? 4/globals.zoom : 1/globals.zoom; - drawRect(walker.right_arm.hand); - globals.ctx.lineWidth = 1/globals.zoom; -} - -drawRect = function(body) { - // set strokestyle and fillstyle before call - globals.ctx.beginPath(); - var fixture = body.GetFixtureList(); - var shape = fixture.GetShape(); - var p0 = body.GetWorldPoint(shape.m_vertices[0]); - globals.ctx.moveTo(p0.x, p0.y); - for(var k = 1; k < 4; k++) { - var p = body.GetWorldPoint(shape.m_vertices[k]); - globals.ctx.lineTo(p.x, p.y); - } - globals.ctx.lineTo(p0.x, p0.y); - - globals.ctx.fill(); - globals.ctx.stroke(); -} - -drawTest = function() { - globals.ctx.strokeStyle = "#000"; - globals.ctx.fillStyle = "#666"; - globals.ctx.lineWidth = 1; - globals.ctx.beginPath(); - globals.ctx.moveTo(0, 0); - globals.ctx.lineTo(0, 2); - globals.ctx.lineTo(2, 2); - - globals.ctx.fill(); - globals.ctx.stroke(); - -} - -getMinMaxDistance = function() { - var min_x = 9999; - var max_x = -1; - var min_y = 9999; - var max_y = -1; - for(var k = 0; k < globals.walkers.length; k++) { - if(globals.walkers[k].health > 0) { - var dist = globals.walkers[k].torso.upper_torso.GetPosition(); - min_x = Math.min(min_x, dist.x); - max_x = Math.max(max_x, dist.x); - min_y = Math.min(min_y, globals.walkers[k].low_foot_height, globals.walkers[k].head_height); - max_y = Math.max(max_y, dist.y); + setFps(fps) { + config.draw_fps = fps; + if(this.draw_interval) + clearInterval(this.draw_interval); + if(fps > 0 && config.simulation_fps > 0) { + this.draw_interval = setInterval(drawFrame, Math.round(1000/config.draw_fps)); } } - return {min_x:min_x, max_x:max_x, min_y:min_y, max_y:max_y}; -} -getZoom = function(min_x, max_x, min_y, max_y) { - var delta_x = Math.abs(max_x - min_x); - var delta_y = Math.abs(max_y - min_y); - var zoom = Math.min(globals.main_screen.width/delta_x,globals.main_screen.height/delta_y); - return zoom; + drawFrame() { + var minmax = getMinMaxDistance(); + this.target_zoom = Math.min(config.max_zoom_factor, getZoom(minmax.min_x, minmax.max_x + 4, minmax.min_y + 2, minmax.max_y + 2.5)); + this.zoom += 0.1*(this.target_zoom - this.zoom); + this.translate_x += 0.1*(1.5-minmax.min_x - this.translate_x); + this.translate_y += 0.3*(minmax.min_y*this.zoom + 280 - this.translate_y); + //this.translate_y = minmax.max_y*this.zoom + 150; + this.ctx.clearRect(0, 0, this.main_screen.width, this.main_screen.height); + this.ctx.save(); + this.ctx.translate(this.translate_x*this.zoom, this.translate_y); + this.ctx.scale(this.zoom, -this.zoom); + drawFloor(); + for(var k = config.population_size - 1; k >= 0 ; k--) { + drawWalker(this.walkers[k]); + } + this.ctx.restore(); + } + + drawFloor() { + this.ctx.strokeStyle = "#444"; + this.ctx.lineWidth = 1/this.zoom; + this.ctx.beginPath(); + var floor_fixture = this.floor.GetFixtureList(); + this.ctx.moveTo(floor_fixture.m_shape.m_vertices[0].x, floor_fixture.m_shape.m_vertices[0].y); + for(var k = 1; k < floor_fixture.m_shape.m_vertices.length; k++) { + this.ctx.lineTo(floor_fixture.m_shape.m_vertices[k].x, floor_fixture.m_shape.m_vertices[k].y); + } + this.ctx.stroke(); + } + + drawWalker (walker) { + var hue = walker.hue || 240 + this.ctx.strokeStyle = "hsl(" + hue + ",100%,0%)"; + this.ctx.fillStyle = "hsl("+hue+",45%,"+(100-15*walker.health/config.walker_health)+"%)"; + this.ctx.lineWidth = 1/this.zoom; + + // left legs and arms first + drawRect(walker.left_leg.lower_leg); + drawRect(walker.left_leg.upper_leg); + drawRect(walker.left_arm.upper_arm); + drawRect(walker.left_arm.lower_arm); + + this.ctx.lineWidth = walker.left_leg.frictionJoint.maxForce? 4/this.zoom : 1/this.zoom; + drawRect(walker.left_leg.foot); + this.ctx.lineWidth = 1/this.zoom; + + this.ctx.lineWidth = walker.left_arm.frictionJoint.maxForce? 4/this.zoom : 1/this.zoom; + drawRect(walker.left_arm.hand); + this.ctx.lineWidth = 1/this.zoom; + + // head + drawRect(walker.head.neck); + drawRect(walker.head.head); + + // torso + drawRect(walker.torso.lower_torso); + drawRect(walker.torso.upper_torso); + + // right legs and arms + drawRect(walker.right_leg.upper_leg); + drawRect(walker.right_leg.lower_leg); + drawRect(walker.right_arm.upper_arm); + drawRect(walker.right_arm.lower_arm); + + this.ctx.lineWidth = walker.right_leg.frictionJoint.maxForce? 4/this.zoom : 1/this.zoom; + drawRect(walker.right_leg.foot); + this.ctx.lineWidth = 1/this.zoom; + this.ctx.lineWidth = walker.right_arm.frictionJoint.maxForce? 4/this.zoom : 1/this.zoom; + drawRect(walker.right_arm.hand); + this.ctx.lineWidth = 1/this.zoom; + } + + drawRect(body) { + // set strokestyle and fillstyle before call + this.ctx.beginPath(); + var fixture = body.GetFixtureList(); + var shape = fixture.GetShape(); + var p0 = body.GetWorldPoint(shape.m_vertices[0]); + this.ctx.moveTo(p0.x, p0.y); + for(var k = 1; k < 4; k++) { + var p = body.GetWorldPoint(shape.m_vertices[k]); + this.ctx.lineTo(p.x, p.y); + } + this.ctx.lineTo(p0.x, p0.y); + + this.ctx.fill(); + this.ctx.stroke(); + } + + drawTest() { + this.ctx.strokeStyle = "#000"; + this.ctx.fillStyle = "#666"; + this.ctx.lineWidth = 1;1 + this.ctx.beginPath(); + this.ctx.moveTo(0, 0); + this.ctx.lineTo(0, 2); + this.ctx.lineTo(2, 2); + + this.ctx.fill(); + this.ctx.stroke(); + + } + + getMinMaxDistance() { + var min_x = 9999; + var max_x = -1; + var min_y = 9999; + var max_y = -1; + for(var k = 0; k < this.walkers.length; k++) { + if(this.walkers[k].health > 0) { + var dist = this.walkers[k].torso.upper_torso.GetPosition(); + min_x = Math.min(min_x, dist.x); + max_x = Math.max(max_x, dist.x); + min_y = Math.min(min_y, this.walkers[k].low_foot_height, this.walkers[k].head_height); + max_y = Math.max(max_y, dist.y); + } + } + return {min_x:min_x, max_x:max_x, min_y:min_y, max_y:max_y}; + } + + getZoom(min_x, max_x, min_y, max_y) { + var delta_x = Math.abs(max_x - min_x); + var delta_y = Math.abs(max_y - min_y); + var zoom = Math.min(this.main_screen.width/delta_x,this.main_screen.height/delta_y); + return zoom; + } } +module.export = {Renderer} diff --git a/js/floor.js b/js/floor.js index 3dce45c..503363e 100644 --- a/js/floor.js +++ b/js/floor.js @@ -1,4 +1,6 @@ -function createFloor(world) { +var b2 = require('../vendor/jsbox2d') + +function createFloor(world, max_floor_tiles) { var body_def = new b2.BodyDef(); var body = world.CreateBody(body_def); body.SetUserData('floor') @@ -13,8 +15,8 @@ function createFloor(world) { new b2.Vec2(2.5, -0.16) ]; - for(var k = 2; k < config.max_floor_tiles; k++) { - var ratio = k / config.max_floor_tiles; + for(var k = 2; k < max_floor_tiles; k++) { + var ratio = k / max_floor_tiles; // add uneven floor by continuing from the last point, plus some random jittering edges.push(new b2.Vec2( edges[edges.length - 1].x + (1 + ratio * Math.random() - ratio / 2), @@ -26,3 +28,5 @@ function createFloor(world) { body.CreateFixture(fix_def); return body; } + +module.exports=createFloor diff --git a/js/walker.js b/js/walker.js index 8dd3468..8386ae9 100644 --- a/js/walker.js +++ b/js/walker.js @@ -1,21 +1,22 @@ // walker has fixed shapes and structures // shape definitions are in the constructor +var randf = (low, high) => Math.random() * (high - low) + low - -function deg2rad(deg) { - return deg/180*Math.PI +function deg2rad(deg) { + return deg / 180 * Math.PI } const STRENGTH = 3 -var Walker = function() { +var Walker = function () { this.__constructor.apply(this, arguments); } -Walker.prototype.__constructor = function(world, floor) { +Walker.prototype.__constructor = function (world, floor, config) { this.world = world; this.floor = floor + this.config = config this.density = 106.2; // common for all fixtures, no reason to be too specific @@ -25,13 +26,18 @@ Walker.prototype.__constructor = function(world, floor) { this.low_foot_height = 0; this.head_height = 0; this.steps = 0; - this.distance = 0 + this.distance = 0 this.last_left_left_forward = true - this.hue = Math.randf(200,360) + this.hue = randf(200, 360) - this.bd = new b2.BodyDef({positions: {x:10, y:-10}}); - this.bd.position.x += Math.randf(-10, 10) + this.bd = new b2.BodyDef({ + positions: { + x: 10, + y: -10 + } + }); + this.bd.position.x += randf(-10, 10) this.bd.type = b2.Body.b2_dynamicBody; this.bd.linearDamping = 0; this.bd.angularDamping = 20; // decay in force/ air friction @@ -99,25 +105,24 @@ Walker.prototype.__constructor = function(world, floor) { // add grip // we don't have data on external forces, so I will just punish for contact with the ground - http://blog.sethladd.com/2011/09/box2d-collision-damage-for-javascript.html - // However we could use a listener or calc force - var self = this + http: //blog.sethladd.com/2011/09/box2d-collision-damage-for-javascript.html + // However we could use a listener or calc force + var self = this this.contactListener = new b2.ContactListener() this.contactListener.BeginContact = function (contact, impulse) { if (contact.m_fixtureA.m_body.m_userData == "floor" | contact.m_fixtureB.m_body.m_userData) { var otherFixture = contact.m_fixtureA.m_body.m_userData == "floor" ? contact.m_fixtureB : contact.m_fixtureA - if (otherFixture.m_body === self.right_leg.foot) { + if (otherFixture.m_body === self.right_leg.foot) { // TODO let the agent act to grip or not. Only if palm or foot down? self.right_leg.frictionJoint.maxForce = 1000 * self.grips[0] self.right_leg.frictionJoint.maxTorque = 1000 * self.grips[0] } else if (otherFixture.m_body === self.left_leg.foot) { self.left_leg.frictionJoint.maxForce = 1000 * self.grips[1] self.left_leg.frictionJoint.maxTorque = 1000 * self.grips[1] - } else if (otherFixture.m_body == self.right_arm.hand) { - // TODO let the agent act to grip or not + } else if (otherFixture.m_body == self.right_arm.hand) { self.right_arm.frictionJoint.maxForce = 1000 * self.grips[2] self.right_arm.frictionJoint.maxTorque = 1000 * self.grips[2] - } else if (otherFixture.m_body === self.left_arm.hand) { + } else if (otherFixture.m_body === self.left_arm.hand) { self.left_arm.frictionJoint.maxForce = 1000 * self.grips[3] self.left_arm.frictionJoint.maxTorque = 1000 * self.grips[3] } @@ -126,29 +131,34 @@ Walker.prototype.__constructor = function(world, floor) { this.contactListener.EndContact = function (contact, impulse) { if (contact.m_fixtureA.m_body.m_userData == "floor" | contact.m_fixtureB.m_body.m_userData) { var otherFixture = contact.m_fixtureA.m_body.m_userData == "floor" ? contact.m_fixtureB : contact.m_fixtureA - if (otherFixture.m_body.m_userData == "right_foot") { - // console.log('grip off ' + otherFixture.m_body.m_userData) - // TODO let the agent act to grip or not - setTimeout(() => { self.right_leg.frictionJoint.maxForce = 0 }, 100) - setTimeout(() => { self.right_leg.frictionJoint.maxTorque = 0 }, 100) - } else if (otherFixture.m_body.m_userData == "left_foot") { - // console.log('grip off ' + otherFixture.m_body.m_userData) + if (otherFixture.m_body === self.right_leg.foot) { + setTimeout(() => { + self.right_leg.frictionJoint.maxForce = 0 + }, 100) + setTimeout(() => { + self.right_leg.frictionJoint.maxTorque = 0 + }, 100) + } else if (otherFixture.m_body === self.left_leg.foot) { self.left_leg.frictionJoint.maxForce = 0 self.left_leg.frictionJoint.maxTorque = 0 - } else if (otherFixture.m_body.m_userData == "right_hand") { - // console.log('grip off ' + otherFixture.m_body.m_userData) - // TODO let the agent act to grip or not - setTimeout(() => { self.right_arm.frictionJoint.maxForce = 0 }, 100) - setTimeout(() => { self.right_arm.frictionJoint.maxTorque = 0 }, 100) - } else if (otherFixture.m_body.m_userData == "left_hand") { - // console.log('grip off ' + otherFixture.m_body.m_userData) + } else if (otherFixture.m_body == self.right_arm.hand) { + setTimeout(() => { + self.right_arm.frictionJoint.maxForce = 0 + }, 100) + setTimeout(() => { + self.right_arm.frictionJoint.maxTorque = 0 + }, 100) + } else if (otherFixture.m_body === self.left_arm.hand) { self.left_arm.frictionJoint.maxForce = 0 self.left_arm.frictionJoint.maxTorque = 0 } - } + } } this.world.SetContactListener(this.contactListener) + // if (typeof document!==undefined) + // this.renderer = new Renderer() + } // Walker.prototype.destroy = function () { @@ -158,21 +168,21 @@ Walker.prototype.__constructor = function(world, floor) { // this.otherJoints.map(joint => this.world.DestroyJoint(joint)) // } -Walker.prototype.createTorso = function() { +Walker.prototype.createTorso = function () { // upper torso - this.bd.position.Set(0.5 - this.leg_def.foot_length/2 + this.leg_def.tibia_width/2, this.leg_def.foot_height/2 + this.leg_def.foot_height/2 + this.leg_def.tibia_length + this.leg_def.femur_length + this.torso_def.lower_height + this.torso_def.upper_height/2); + this.bd.position.Set(0.5 - this.leg_def.foot_length / 2 + this.leg_def.tibia_width / 2, this.leg_def.foot_height / 2 + this.leg_def.foot_height / 2 + this.leg_def.tibia_length + this.leg_def.femur_length + this.torso_def.lower_height + this.torso_def.upper_height / 2); var upper_torso = this.world.CreateBody(this.bd); upper_torso.SetUserData('upper_torso') - this.fd.shape.SetAsBox(this.torso_def.upper_width/2, this.torso_def.upper_height/2); + this.fd.shape.SetAsBox(this.torso_def.upper_width / 2, this.torso_def.upper_height / 2); upper_torso.CreateFixture(this.fd); // lower torso - this.bd.position.Set(0.5 - this.leg_def.foot_length/2 + this.leg_def.tibia_width/2, this.leg_def.foot_height/2 + this.leg_def.foot_height/2 + this.leg_def.tibia_length + this.leg_def.femur_length + this.torso_def.lower_height/2); + this.bd.position.Set(0.5 - this.leg_def.foot_length / 2 + this.leg_def.tibia_width / 2, this.leg_def.foot_height / 2 + this.leg_def.foot_height / 2 + this.leg_def.tibia_length + this.leg_def.femur_length + this.torso_def.lower_height / 2); var lower_torso = this.world.CreateBody(this.bd); - lower_torso.SetUserData('lower_torso' ) + lower_torso.SetUserData('lower_torso') - this.fd.shape.SetAsBox(this.torso_def.lower_width/2, this.torso_def.lower_height/2); + this.fd.shape.SetAsBox(this.torso_def.lower_width / 2, this.torso_def.lower_height / 2); lower_torso.CreateFixture(this.fd); // torso joint @@ -180,53 +190,56 @@ Walker.prototype.createTorso = function() { // For 3d definition https://github.com/openai/gym/blob/master/gym/envs/mujoco/assets/humanoid.xml var jd = new b2.RevoluteJointDef(); var position = upper_torso.GetPosition().Clone(); - position.y -= this.torso_def.upper_height/2; - position.x -= this.torso_def.lower_width/3; + position.y -= this.torso_def.upper_height / 2; + position.x -= this.torso_def.lower_width / 3; jd.Initialize(upper_torso, lower_torso, position); - jd.lowerAngle = deg2rad(-75/2); + jd.lowerAngle = deg2rad(-75 / 2); jd.upperAngle = deg2rad(30 / 2); jd.enableLimit = true; jd.maxMotorTorque = 150 * STRENGTH; jd.motorSpeed = 0; jd.enableMotor = true; var j = this.world.CreateJoint(jd) - j.SetUserData('torso_joint' ) + j.SetUserData('torso_joint') this.joints.push(j); - return {upper_torso: upper_torso, lower_torso: lower_torso}; + return { + upper_torso: upper_torso, + lower_torso: lower_torso + }; } -Walker.prototype.createLeg = function(label) { +Walker.prototype.createLeg = function (label) { // upper leg - this.bd.position.Set(0.5 - this.leg_def.foot_length/2 + this.leg_def.tibia_width/2, this.leg_def.foot_height/2 + this.leg_def.foot_height/2 + this.leg_def.tibia_length + this.leg_def.femur_length/2); + this.bd.position.Set(0.5 - this.leg_def.foot_length / 2 + this.leg_def.tibia_width / 2, this.leg_def.foot_height / 2 + this.leg_def.foot_height / 2 + this.leg_def.tibia_length + this.leg_def.femur_length / 2); var upper_leg = this.world.CreateBody(this.bd); - this.fd.shape.SetAsBox(this.leg_def.femur_width/2, this.leg_def.femur_length/2); + this.fd.shape.SetAsBox(this.leg_def.femur_width / 2, this.leg_def.femur_length / 2); upper_leg.CreateFixture(this.fd); upper_leg.SetUserData(label + 'upper_leg') // lower leg - this.bd.position.Set(0.5 - this.leg_def.foot_length/2 + this.leg_def.tibia_width/2, this.leg_def.foot_height/2 + this.leg_def.foot_height/2 + this.leg_def.tibia_length/2); + this.bd.position.Set(0.5 - this.leg_def.foot_length / 2 + this.leg_def.tibia_width / 2, this.leg_def.foot_height / 2 + this.leg_def.foot_height / 2 + this.leg_def.tibia_length / 2); var lower_leg = this.world.CreateBody(this.bd); - this.fd.shape.SetAsBox(this.leg_def.tibia_width/2, this.leg_def.tibia_length/2); + this.fd.shape.SetAsBox(this.leg_def.tibia_width / 2, this.leg_def.tibia_length / 2); lower_leg.CreateFixture(this.fd); lower_leg.SetUserData(label + 'lower_leg') // foot - this.bd.position.Set(0.5, this.leg_def.foot_height/2); + this.bd.position.Set(0.5, this.leg_def.foot_height / 2); var foot = this.world.CreateBody(this.bd); - this.fd.shape.SetAsBox(this.leg_def.foot_length/2, this.leg_def.foot_height/2); + this.fd.shape.SetAsBox(this.leg_def.foot_length / 2, this.leg_def.foot_height / 2); foot.CreateFixture(this.fd); foot.SetUserData(label + 'foot') var fjd = new b2.FrictionJointDef(); - var position = new b2.Vec2(0,0) + var position = new b2.Vec2(0, 0) fjd.Initialize(foot, this.floor, position) fjd.maxForce = 0; //This the most force the joint will apply to your object. The faster its moving the more force applied fjd.maxTorque = 0; //Set to 0 to prevent rotation - fjd.userData = label+'foot_friction_joint' + fjd.userData = label + 'foot_friction_joint' fjd.collideConnected = true var frictionJoint = this.world.CreateJoint(fjd) this.otherJoints.push(frictionJoint) @@ -234,8 +247,8 @@ Walker.prototype.createLeg = function(label) { // leg joints var jd = new b2.RevoluteJointDef(); var position = upper_leg.GetPosition().Clone(); - position.y -= this.leg_def.femur_length/2; - position.x += this.leg_def.femur_width/4; + position.y -= this.leg_def.femur_length / 2; + position.x += this.leg_def.femur_width / 4; jd.Initialize(upper_leg, lower_leg, position); jd.lowerAngle = deg2rad(-100); jd.upperAngle = deg2rad(-2); @@ -250,7 +263,7 @@ Walker.prototype.createLeg = function(label) { // foot joint var jd = new b2.RevoluteJointDef(); var position = lower_leg.GetPosition().Clone(); - position.y -= this.leg_def.tibia_length/2; + position.y -= this.leg_def.tibia_length / 2; jd.Initialize(lower_leg, foot, position); jd.lowerAngle = -deg2rad(-36); jd.upperAngle = deg2rad(30); @@ -262,41 +275,46 @@ Walker.prototype.createLeg = function(label) { j.SetUserData(label + 'foot_joint') this.joints.push(j); - return {upper_leg: upper_leg, lower_leg: lower_leg, foot:foot, frictionJoint:frictionJoint}; + return { + upper_leg: upper_leg, + lower_leg: lower_leg, + foot: foot, + frictionJoint: frictionJoint + }; } -Walker.prototype.createArm = function(label) { +Walker.prototype.createArm = function (label) { // upper arm - this.bd.position.Set(0.5 - this.leg_def.foot_length/2 + this.leg_def.tibia_width/2, this.leg_def.foot_height/2 + this.leg_def.foot_height/2 + this.leg_def.tibia_length + this.leg_def.femur_length + this.torso_def.lower_height + this.torso_def.upper_height - this.arm_def.arm_length/2); + this.bd.position.Set(0.5 - this.leg_def.foot_length / 2 + this.leg_def.tibia_width / 2, this.leg_def.foot_height / 2 + this.leg_def.foot_height / 2 + this.leg_def.tibia_length + this.leg_def.femur_length + this.torso_def.lower_height + this.torso_def.upper_height - this.arm_def.arm_length / 2); var upper_arm = this.world.CreateBody(this.bd); - this.fd.shape.SetAsBox(this.arm_def.arm_width/2, this.arm_def.arm_length/2); + this.fd.shape.SetAsBox(this.arm_def.arm_width / 2, this.arm_def.arm_length / 2); upper_arm.CreateFixture(this.fd); upper_arm.SetUserData(label + 'upper_arm') // lower arm - this.bd.position.Set(0.5 - this.leg_def.foot_length/2 + this.leg_def.tibia_width/2, this.leg_def.foot_height/2 + this.leg_def.foot_height/2 + this.leg_def.tibia_length + this.leg_def.femur_length + this.torso_def.lower_height + this.torso_def.upper_height - this.arm_def.arm_length - this.arm_def.forearm_length/2); + this.bd.position.Set(0.5 - this.leg_def.foot_length / 2 + this.leg_def.tibia_width / 2, this.leg_def.foot_height / 2 + this.leg_def.foot_height / 2 + this.leg_def.tibia_length + this.leg_def.femur_length + this.torso_def.lower_height + this.torso_def.upper_height - this.arm_def.arm_length - this.arm_def.forearm_length / 2); var lower_arm = this.world.CreateBody(this.bd); - this.fd.shape.SetAsBox(this.arm_def.forearm_width/2, this.arm_def.forearm_length/2); + this.fd.shape.SetAsBox(this.arm_def.forearm_width / 2, this.arm_def.forearm_length / 2); lower_arm.CreateFixture(this.fd); lower_arm.SetUserData(label + 'lower_arm') // hand - this.bd.position.Set(0.5 - this.leg_def.foot_length/2 + this.leg_def.tibia_width/2, this.leg_def.foot_height/2 + this.leg_def.foot_height/2 + this.leg_def.tibia_length + this.leg_def.femur_length + this.torso_def.lower_height + this.torso_def.upper_height - this.arm_def.arm_length - this.arm_def.forearm_length - this.arm_def.hand_length/2); + this.bd.position.Set(0.5 - this.leg_def.foot_length / 2 + this.leg_def.tibia_width / 2, this.leg_def.foot_height / 2 + this.leg_def.foot_height / 2 + this.leg_def.tibia_length + this.leg_def.femur_length + this.torso_def.lower_height + this.torso_def.upper_height - this.arm_def.arm_length - this.arm_def.forearm_length - this.arm_def.hand_length / 2); var hand = this.world.CreateBody(this.bd); - this.fd.shape.SetAsBox(this.arm_def.hand_height/2, this.arm_def.hand_length/2); + this.fd.shape.SetAsBox(this.arm_def.hand_height / 2, this.arm_def.hand_length / 2); hand.CreateFixture(this.fd); hand.SetUserData(label + 'hand') var fjd = new b2.FrictionJointDef(); - var position = new b2.Vec2(0,0) + var position = new b2.Vec2(0, 0) fjd.Initialize(hand, this.floor, position) fjd.maxForce = 0; //This the most force the joint will apply to your object. The faster its moving the more force applied fjd.maxTorque = 0; //Set to 0 to prevent rotation - fjd.userData = label+'hand_friction_joint' + fjd.userData = label + 'hand_friction_joint' fjd.collideConnected = true var frictionJoint = this.world.CreateJoint(fjd) this.otherJoints.push(frictionJoint) @@ -305,7 +323,7 @@ Walker.prototype.createArm = function(label) { // arm join var jd = new b2.RevoluteJointDef(); var position = upper_arm.GetPosition().Clone(); - position.y -= this.arm_def.arm_length/2; + position.y -= this.arm_def.arm_length / 2; jd.Initialize(upper_arm, lower_arm, position); jd.lowerAngle = deg2rad(0); jd.upperAngle = deg2rad(85); @@ -320,7 +338,7 @@ Walker.prototype.createArm = function(label) { // hand joint var jd = new b2.RevoluteJointDef(); var position = lower_arm.GetPosition().Clone(); - position.y -= this.arm_def.forearm_length/2; + position.y -= this.arm_def.forearm_length / 2; jd.Initialize(lower_arm, hand, position); jd.lowerAngle = deg2rad(-35); jd.upperAngle = deg2rad(35); @@ -333,33 +351,34 @@ Walker.prototype.createArm = function(label) { this.joints.push(j); return { - upper_arm: upper_arm, lower_arm: lower_arm, + upper_arm: upper_arm, + lower_arm: lower_arm, hand: hand, - frictionJoint:frictionJoint + frictionJoint: frictionJoint }; } -Walker.prototype.createHead = function() { +Walker.prototype.createHead = function () { // neck - this.bd.position.Set(0.5 - this.leg_def.foot_length/2 + this.leg_def.tibia_width/2, this.leg_def.foot_height/2 + this.leg_def.foot_height/2 + this.leg_def.tibia_length + this.leg_def.femur_length + this.torso_def.lower_height + this.torso_def.upper_height + this.head_def.neck_height/2); + this.bd.position.Set(0.5 - this.leg_def.foot_length / 2 + this.leg_def.tibia_width / 2, this.leg_def.foot_height / 2 + this.leg_def.foot_height / 2 + this.leg_def.tibia_length + this.leg_def.femur_length + this.torso_def.lower_height + this.torso_def.upper_height + this.head_def.neck_height / 2); var neck = this.world.CreateBody(this.bd); - - this.fd.shape.SetAsBox(this.head_def.neck_width/2, this.head_def.neck_height/2); + + this.fd.shape.SetAsBox(this.head_def.neck_width / 2, this.head_def.neck_height / 2); neck.CreateFixture(this.fd); neck.SetUserData('neck') // head - this.bd.position.Set(0.5 - this.leg_def.foot_length/2 + this.leg_def.tibia_width/2, this.leg_def.foot_height/2 + this.leg_def.foot_height/2 + this.leg_def.tibia_length + this.leg_def.femur_length + this.torso_def.lower_height + this.torso_def.upper_height + this.head_def.neck_height + this.head_def.head_height/2); + this.bd.position.Set(0.5 - this.leg_def.foot_length / 2 + this.leg_def.tibia_width / 2, this.leg_def.foot_height / 2 + this.leg_def.foot_height / 2 + this.leg_def.tibia_length + this.leg_def.femur_length + this.torso_def.lower_height + this.torso_def.upper_height + this.head_def.neck_height + this.head_def.head_height / 2); var head = this.world.CreateBody(this.bd); - this.fd.shape.SetAsBox(this.head_def.head_width/2, this.head_def.head_height/2); + this.fd.shape.SetAsBox(this.head_def.head_width / 2, this.head_def.head_height / 2); head.CreateFixture(this.fd); head.SetUserData('head') // neck joint var jd = new b2.RevoluteJointDef(); var position = neck.GetPosition().Clone(); - position.y += this.head_def.neck_height/2; + position.y += this.head_def.neck_height / 2; jd.Initialize(head, neck, position); jd.lowerAngle = -0.1; jd.upperAngle = 0.2; @@ -371,17 +390,20 @@ Walker.prototype.createHead = function() { j.SetUserData('neck_joint') this.joints.push(j); - return {head: head, neck: neck}; + return { + head: head, + neck: neck + }; } -Walker.prototype.connectParts = function() { +Walker.prototype.connectParts = function () { //neck/torso var jd = new b2.WeldJointDef(); jd.bodyA = this.head.neck; jd.bodyB = this.torso.upper_torso; - jd.localAnchorA = new b2.Vec2(0, -this.head_def.neck_height/2); - jd.localAnchorB = new b2.Vec2(0, this.torso_def.upper_height/2); + jd.localAnchorA = new b2.Vec2(0, -this.head_def.neck_height / 2); + jd.localAnchorB = new b2.Vec2(0, this.torso_def.upper_height / 2); jd.referenceAngle = 0; var j = this.world.CreateJoint(jd); j.SetUserData('neck-joint') @@ -390,7 +412,7 @@ Walker.prototype.connectParts = function() { // torso/arms var jd = new b2.RevoluteJointDef(); position = this.torso.upper_torso.GetPosition().Clone(); - position.y += this.torso_def.upper_height/2; + position.y += this.torso_def.upper_height / 2; jd.Initialize(this.torso.upper_torso, this.right_arm.upper_arm, position); jd.lowerAngle = deg2rad(-60); jd.upperAngle = deg2rad(125); @@ -417,7 +439,7 @@ Walker.prototype.connectParts = function() { // torso/legs var jd = new b2.RevoluteJointDef(); position = this.torso.lower_torso.GetPosition().Clone(); - position.y -= this.torso_def.lower_height/2; + position.y -= this.torso_def.lower_height / 2; jd.Initialize(this.torso.lower_torso, this.right_leg.upper_leg, position); jd.lowerAngle = deg2rad(-10); jd.upperAngle = deg2rad(80); @@ -442,7 +464,7 @@ Walker.prototype.connectParts = function() { this.joints.push(j); } -Walker.prototype.getBodies = function() { +Walker.prototype.getBodies = function () { return [ this.head.head, @@ -464,60 +486,66 @@ Walker.prototype.getBodies = function() { ]; } -Walker.prototype.randomise = function (n) { +Walker.prototype.randomise = function (n) { for (var k = 0; k < this.joints.length; k++) { - this.joints[k].SetMotorSpeed(Math.randf(-n, n)) + this.joints[k].SetMotorSpeed(randf(-n, n)) } } Walker.prototype.getState = function () { - var self = this; + var self = this; var state = [] this.bodies - .forEach((body) => { + .forEach((body) => { // see http://www.box2dflash.org/docs/2.0.2/reference/Box2D/Dynamics/b2Body.html#GetLocalVector() var t = body.GetTransform() // world transform of the body's origin. state.push(t.p.x) // world transform of the body's origin. state.push(t.p.y) // world transform of the body's origin. state.push(t.q.s) // world transform of the body's origin. state.push(t.q.c) // world transform of the body's origin. - + var dt = body.GetLinearVelocity() // Get the linear velocity of the center of mass (world). state.push(dt.x) state.push(dt.y) - var lp = self.torso.upper_torso.GetLocalPoint(body.GetWorldCenter())// Get the bodypart position relative to the upper torso - state.push(lp.x) + var lp = self.torso.upper_torso.GetLocalPoint(body.GetWorldCenter()) // Get the bodypart position relative to the upper torso + state.push(lp.x) state.push(lp.y) state.push(body.GetAngularVelocity()) // the angular velocity in radians/second. - state.push(body.GetAngle()) // the current world rotation angle in radians. + state.push(body.GetAngle()) // the current world rotation angle in radians. }, []) - this.joints.forEach(joint => { + this.joints.forEach(joint => { // http://www.box2dflash.org/docs/2.0.2/reference/Box2D/Dynamics/Joints/b2RevoluteJoint.html state.push(joint.GetJointAngle()) // Get the current joint angle in radians. state.push(joint.GetJointSpeed()) // Get the current joint angle speed in radians per second - state.push(joint.GetMotorSpeed()) + state.push(joint.GetMotorSpeed()) }) return state } -Walker.prototype.simulationPreStep = function (motorSpeeds) { +Walker.prototype.simulationPreStep = function (motorSpeeds) { // act for (var k = 0; k < this.joints.length; k++) { this.joints[k].SetMotorSpeed(motorSpeeds[k] * 10); // action can range from -3 to 3, radians per second } - for (let i = 0; i < motorSpeeds.length-this.joints.length; i++) { + for (let i = 0; i < motorSpeeds.length - this.joints.length; i++) { this.grips[i] = motorSpeeds[i] > 0 } if (motorSpeeds[0] <= 0) this.right_leg.frictionJoint.maxForce = this.right_leg.frictionJoint.maxTorque if (motorSpeeds[1] <= 0) this.left_leg.frictionJoint.maxForce = this.left_leg.frictionJoint.maxTorque if (motorSpeeds[2] <= 0) this.left_arm.frictionJoint.maxForce = this.left_arm.frictionJoint.maxTorque if (motorSpeeds[3] <= 0) this.right_arm.frictionJoint.maxForce = this.right_arm.frictionJoint.maxTorque - // TODO turn of grups } -Walker.prototype.simulationStep = function (motorSpeeds) { +Walker.prototype.step = function (motorSpeeds) { + /* + Take one step into the environement + @delta (Float) time since the last update + @action: (Integer) The action to take (can be null if no action) + */ + this.simulationPreStep(motorSpeeds) + this.world.Step(1 / this.config.time_step, this.config.velocity_iterations, this.config.position_iterations); this.steps++ /* score/reward */ // reward copied from OpenAI Gym Humanoid Walker https://github.com/openai/gym/blob/master/gym/envs/mujoco/humanoid.py @@ -525,9 +553,9 @@ Walker.prototype.simulationStep = function (motorSpeeds) { // https://github.com/openai/gym/blob/master/gym/envs/mujoco/assets/humanoidstandup.xml // reward for keeping head up, compared to feet - var mean_foot_height = (this.left_leg.foot.GetPosition().y + this.right_leg.foot.GetPosition().y)/2 - - var head_height_reward = (this.head.head.GetPosition().y - mean_foot_height)* 400; // it's head should be above it's feet 2*(-0.25-2) + var mean_foot_height = (this.left_leg.foot.GetPosition().y + this.right_leg.foot.GetPosition().y) / 2 + + var head_height_reward = (this.head.head.GetPosition().y - mean_foot_height) * 400; // it's head should be above it's feet 2*(-0.25-2) // reward for moving one leg beyond the other (stepping) var left_leg_forward = this.right_leg.foot.GetPosition().x > this.left_leg.foot.GetPosition().x; @@ -536,7 +564,7 @@ Walker.prototype.simulationStep = function (motorSpeeds) { // cost for moving joints to unnatural positions (fraction of movement range in the relevant direction) var jointFractionMovement = j => j.GetJointAngle() > 0 ? j.GetJointAngle() / (j.GetUpperLimit() + 1) : j.GetJointAngle() / (j.GetLowerLimit() - 1) - var quad_joint_angle_cost = - 0.10 * this.joints.map(j => jointFractionMovement(j) * 1.2) + var quad_joint_angle_cost = -0.10 * this.joints.map(j => jointFractionMovement(j) * 1.2) .reduce((o, v) => o + v * v, 0) quad_joint_angle_cost = Math.max(quad_joint_angle_cost, -10) @@ -552,19 +580,10 @@ Walker.prototype.simulationStep = function (motorSpeeds) { quad_power_cost = Math.max(quad_power_cost, -10) // Lets be nice, all entities should find overall happiness in what they do - var bonus_happiness = 5 + var bonus_happiness = 5 - // we don't have data on external forces, so I will just punish for contact with the ground - http://blog.sethladd.com/2011/09/box2d-collision-damage-for-javascript.html - // However we could use a listener or calc force - // var listener = new b2.ContactListener() - // listener.PostSolve = function (contact, impulse) { - // if (contact) console.log(contact) - // } - // this.world.SetContactListener(listener) - // } var contacts = this.bodies.map(b => b.GetContactList()).filter(b => b).length - quad_contact_cost = -Math.min(contacts - 4, 10)/2 + quad_contact_cost = -Math.min(contacts - 4, 10) / 2 this.rewards = { lin_vel_reward, @@ -575,15 +594,90 @@ Walker.prototype.simulationStep = function (motorSpeeds) { head_height_reward, leg_switch_reward } - - this.reward = Object.values(this.rewards).reduce((tot,v)=>tot+v, 0)/3 + + this.reward = Object.values(this.rewards).reduce((tot, v) => tot + v, 0) / 3 var info = { episodeSteps: this.steps, - reward:this.reward, + reward: this.reward, position, ...this.rewards } var done = 0 + this.world.ClearForces(); + console.debug('reward', this.rewards) return [this.getState(), this.reward, done, info] } + +Walker.prototype.getLastReward = function () { + return this.reward +} + +Walker.prototype.render = function (val) { + if (val) { + this.renderer.setFps(this.config.draw_fps) + this.steping = true; + } else { + this.renderer.setFps(0) + this.steping = false; + } +} +Walker.prototype.reset = function () { + /** Reset position to initial or random position TODO */ + console.log('reset not implemented') +} +Walker.prototype.shuffle = function () { + /** Reset position to initial or random position TODO */ + console.log('shuffle not implemented') +} + + +config = { + time_step: 60, + simulation_fps: 60, + draw_fps: 60, + velocity_iterations: 8, + position_iterations: 3, + max_zoom_factor: 130, + min_motor_speed: -2, + max_motor_speed: 2, + population_size: 1, + walker_health: 100, + max_floor_tiles: 50, + round_length: 1000, + min_body_delta: 0, + min_leg_delta: 0.0, +}; + +var b2 = require('../vendor/jsbox2d') +var createFloor = require('./floor.js') +var DDPGAgent = require('./ddpg/ddpg_agent') +var DDPGAgent = require('./ddpg/ddpg_agent') + + +var world = new b2.World(new b2.Vec2(0, -10)) +floor = createFloor(world, config.max_floor_tiles); +var env = new Walker(world, floor, config) + +var nbActions = env.joints.length + 4 +var stateSize = env.bodies.length * 10 + env.joints.length * 3 + +var agent = new DDPGAgent(env, { + stateSize, + nbActions, + resetEpisode: true, + desiredActionStddev: 0.4, + initialStddev: 0.4, + actorFirstLayerSize: 128, + actorSecondLayerSize: 64, + criticFirstLayerSSize: 128, + criticFirstLayerASize: 128, + criticSecondLayerSize: 64, + nbEpochs: 1000 +}); +agent.train(true); + +// setInterval(() => { +// walker.step() +// console.log(walker.getState()) +// }, 100) diff --git a/package.json b/package.json index 1e5a6e4..997af8d 100644 --- a/package.json +++ b/package.json @@ -3,7 +3,12 @@ "version": "0.0.1", "description": "TODO: Write a project description", "main": "index.js", - "dependencies": {}, + "dependencies": { + "@tensorflow/tfjs": "^0.14.0", + "canvas": "^2.1.0", + "jsdom": "^13.0.0", + "phaser": "^3.15.1" + }, "devDependencies": {}, "scripts": { "test": "echo \"Error: no test specified\" && exit 1"