Hugdir's picture
Upload trained Pyramids PPO RND agent
298c441 verified
Raw
History Blame Contribute Delete
18.8 kB
{
"name": "root",
"gauges": {
"Pyramids.Policy.Entropy.mean": {
"value": 0.23843881487846375,
"min": 0.23753118515014648,
"max": 1.4292476177215576,
"count": 52
},
"Pyramids.Policy.Entropy.sum": {
"value": 7134.08935546875,
"min": 7038.52392578125,
"max": 43357.65625,
"count": 52
},
"Pyramids.Step.mean": {
"value": 1559896.0,
"min": 29952.0,
"max": 1559896.0,
"count": 52
},
"Pyramids.Step.sum": {
"value": 1559896.0,
"min": 29952.0,
"max": 1559896.0,
"count": 52
},
"Pyramids.Policy.ExtrinsicValueEstimate.mean": {
"value": 0.8326560258865356,
"min": -0.09205543994903564,
"max": 0.8420236706733704,
"count": 52
},
"Pyramids.Policy.ExtrinsicValueEstimate.sum": {
"value": 248.13150024414062,
"min": -22.185361862182617,
"max": 257.65924072265625,
"count": 52
},
"Pyramids.Policy.RndValueEstimate.mean": {
"value": 0.010321666486561298,
"min": -0.00941386166960001,
"max": 0.30597904324531555,
"count": 52
},
"Pyramids.Policy.RndValueEstimate.sum": {
"value": 3.0758566856384277,
"min": -2.607639789581299,
"max": 73.74095153808594,
"count": 52
},
"Pyramids.Losses.PolicyLoss.mean": {
"value": 0.07147382796156357,
"min": 0.06486261799195224,
"max": 0.07377905931142677,
"count": 52
},
"Pyramids.Losses.PolicyLoss.sum": {
"value": 1.00063359146189,
"min": 0.48229706238743836,
"max": 1.1066858896714016,
"count": 52
},
"Pyramids.Losses.ValueLoss.mean": {
"value": 0.01327213931332097,
"min": 0.00015669237809535664,
"max": 0.0166132790467921,
"count": 52
},
"Pyramids.Losses.ValueLoss.sum": {
"value": 0.1858099503864936,
"min": 0.0020370009152396364,
"max": 0.2491991857018815,
"count": 52
},
"Pyramids.Policy.LearningRate.mean": {
"value": 0.00014549105864585473,
"min": 0.00014549105864585473,
"max": 0.00029838354339596195,
"count": 52
},
"Pyramids.Policy.LearningRate.sum": {
"value": 0.0020368748210419663,
"min": 0.0020368748210419663,
"max": 0.004027387257537632,
"count": 52
},
"Pyramids.Policy.Epsilon.mean": {
"value": 0.1484970023809524,
"min": 0.1484970023809524,
"max": 0.19946118095238097,
"count": 52
},
"Pyramids.Policy.Epsilon.sum": {
"value": 2.0789580333333335,
"min": 1.3962282666666668,
"max": 2.842462366666667,
"count": 52
},
"Pyramids.Policy.Beta.mean": {
"value": 0.004854850537857143,
"min": 0.004854850537857143,
"max": 0.009946171977142856,
"count": 52
},
"Pyramids.Policy.Beta.sum": {
"value": 0.06796790753000001,
"min": 0.06796790753000001,
"max": 0.13426199043,
"count": 52
},
"Pyramids.Losses.RNDLoss.mean": {
"value": 0.01016409695148468,
"min": 0.009826130233705044,
"max": 0.469675749540329,
"count": 52
},
"Pyramids.Losses.RNDLoss.sum": {
"value": 0.14229735732078552,
"min": 0.13936378061771393,
"max": 3.2877302169799805,
"count": 52
},
"Pyramids.Environment.EpisodeLength.mean": {
"value": 230.04724409448818,
"min": 220.32575757575756,
"max": 999.0,
"count": 52
},
"Pyramids.Environment.EpisodeLength.sum": {
"value": 29216.0,
"min": 15984.0,
"max": 32708.0,
"count": 52
},
"Pyramids.Environment.CumulativeReward.mean": {
"value": 1.7541999853267445,
"min": -1.0000000521540642,
"max": 1.7697099173114499,
"count": 52
},
"Pyramids.Environment.CumulativeReward.sum": {
"value": 222.78339813649654,
"min": -30.99380160868168,
"max": 235.2053981423378,
"count": 52
},
"Pyramids.Policy.ExtrinsicReward.mean": {
"value": 1.7541999853267445,
"min": -1.0000000521540642,
"max": 1.7697099173114499,
"count": 52
},
"Pyramids.Policy.ExtrinsicReward.sum": {
"value": 222.78339813649654,
"min": -30.99380160868168,
"max": 235.2053981423378,
"count": 52
},
"Pyramids.Policy.RndReward.mean": {
"value": 0.024360329406096472,
"min": 0.023964540308591544,
"max": 9.351730020716786,
"count": 52
},
"Pyramids.Policy.RndReward.sum": {
"value": 3.093761834574252,
"min": 3.08454118440568,
"max": 149.62768033146858,
"count": 52
},
"Pyramids.IsTraining.mean": {
"value": 1.0,
"min": 1.0,
"max": 1.0,
"count": 52
},
"Pyramids.IsTraining.sum": {
"value": 1.0,
"min": 1.0,
"max": 1.0,
"count": 52
}
},
"metadata": {
"timer_format_version": "0.1.0",
"start_time_seconds": "1790469955",
"python_version": "3.10.12 (main, Jul 5 2023, 18:54:27) [GCC 11.2.0]",
"command_line_arguments": "/usr/local/miniconda/envs/mlagents/bin/mlagents-learn /content/ml-agents/config/ppo/PyramidsRND.yaml --env=./training-envs-executables/linux/Pyramids/Pyramids --run-id=Pyramids Training --no-graphics",
"mlagents_version": "1.2.0.dev0",
"mlagents_envs_version": "1.2.0.dev0",
"communication_protocol_version": "1.5.0",
"pytorch_version": "2.8.0+cu128",
"numpy_version": "1.23.5",
"end_time_seconds": "1790473716"
},
"total": 3761.398091067,
"count": 1,
"self": 0.4592522090001694,
"children": {
"run_training.setup": {
"total": 0.022923066999965158,
"count": 1,
"self": 0.022923066999965158
},
"TrainerController.start_learning": {
"total": 3760.9159157910003,
"count": 1,
"self": 1.9284425549435582,
"children": {
"TrainerController._reset_env": {
"total": 3.4518364200000633,
"count": 1,
"self": 3.4518364200000633
},
"TrainerController.advance": {
"total": 3755.3513909860562,
"count": 101305,
"self": 2.0638578332041106,
"children": {
"env_step": {
"total": 2801.4038006209926,
"count": 101305,
"self": 2577.6385888391824,
"children": {
"SubprocessEnvManager._take_step": {
"total": 222.5810634649638,
"count": 101305,
"self": 7.046059270903925,
"children": {
"TorchPolicy.evaluate": {
"total": 215.53500419405987,
"count": 98044,
"self": 215.53500419405987
}
}
},
"workers": {
"total": 1.1841483168466311,
"count": 101304,
"self": 0.0,
"children": {
"worker_root": {
"total": 3751.5966338229764,
"count": 101304,
"is_parallel": true,
"self": 1349.8627890279322,
"children": {
"run_training.setup": {
"total": 0.0,
"count": 0,
"is_parallel": true,
"self": 0.0,
"children": {
"steps_from_proto": {
"total": 0.004818004000071596,
"count": 1,
"is_parallel": true,
"self": 0.0035155789994405495,
"children": {
"_process_rank_one_or_two_observation": {
"total": 0.0013024250006310467,
"count": 8,
"is_parallel": true,
"self": 0.0013024250006310467
}
}
},
"UnityEnvironment.step": {
"total": 0.05150600199999644,
"count": 1,
"is_parallel": true,
"self": 0.0006052000001091073,
"children": {
"UnityEnvironment._generate_step_input": {
"total": 0.0005078389999653155,
"count": 1,
"is_parallel": true,
"self": 0.0005078389999653155
},
"communicator.exchange": {
"total": 0.04876108500002374,
"count": 1,
"is_parallel": true,
"self": 0.04876108500002374
},
"steps_from_proto": {
"total": 0.001631877999898279,
"count": 1,
"is_parallel": true,
"self": 0.0003466269997716154,
"children": {
"_process_rank_one_or_two_observation": {
"total": 0.0012852510001266637,
"count": 8,
"is_parallel": true,
"self": 0.0012852510001266637
}
}
}
}
}
}
},
"UnityEnvironment.step": {
"total": 2401.733844795044,
"count": 101303,
"is_parallel": true,
"self": 52.23696876606664,
"children": {
"UnityEnvironment._generate_step_input": {
"total": 34.62122753802191,
"count": 101303,
"is_parallel": true,
"self": 34.62122753802191
},
"communicator.exchange": {
"total": 2149.950263898044,
"count": 101303,
"is_parallel": true,
"self": 2149.950263898044
},
"steps_from_proto": {
"total": 164.92538459291177,
"count": 101303,
"is_parallel": true,
"self": 34.348601567288824,
"children": {
"_process_rank_one_or_two_observation": {
"total": 130.57678302562294,
"count": 810424,
"is_parallel": true,
"self": 130.57678302562294
}
}
}
}
}
}
}
}
}
}
},
"trainer_advance": {
"total": 951.8837325318596,
"count": 101304,
"self": 3.8505665268289704,
"children": {
"process_trajectory": {
"total": 173.1932540710327,
"count": 101304,
"self": 172.84371863903243,
"children": {
"RLTrainer._checkpoint": {
"total": 0.34953543200026616,
"count": 3,
"self": 0.34953543200026616
}
}
},
"_update_policy": {
"total": 774.8399119339979,
"count": 721,
"self": 413.7242994168921,
"children": {
"TorchPPOOptimizer.update": {
"total": 361.1156125171058,
"count": 35757,
"self": 361.1156125171058
}
}
}
}
}
}
},
"trainer_threads": {
"total": 1.4269999155658297e-06,
"count": 1,
"self": 1.4269999155658297e-06
},
"TrainerController._save_models": {
"total": 0.1842444030007755,
"count": 1,
"self": 0.0015035500009616953,
"children": {
"RLTrainer._checkpoint": {
"total": 0.1827408529998138,
"count": 1,
"self": 0.1827408529998138
}
}
}
}
}
}
}