Reinforcement Learning
ml-agents
TensorBoard
ONNX
Pyramids
deep-reinforcement-learning
ML-Agents-Pyramids
Instructions to use nirmanpatel/ppo-PyramidsRND with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- ml-agents
How to use nirmanpatel/ppo-PyramidsRND with ml-agents:
mlagents-load-from-hf --repo-id="nirmanpatel/ppo-PyramidsRND" --local-dir="./downloads"
- Notebooks
- Google Colab
- Kaggle
Download run_logs/timers.json from nirmanpatel/ppo-PyramidsRND: direct link, hf CLI and curl.
- Browser
- Download file 18.8 kB
-
https://huggingface.co/nirmanpatel/ppo-PyramidsRND/resolve/main/run_logs/timers.json
- Command line
-
hf download hf://nirmanpatel/ppo-PyramidsRND/run_logs/timers.json
-
curl -L -o timers.json https://huggingface.co/nirmanpatel/ppo-PyramidsRND/resolve/main/run_logs/timers.json
18.8 kB
| { | |
| "name": "root", | |
| "gauges": { | |
| "Pyramids.Policy.Entropy.mean": { | |
| "value": 0.2367386519908905, | |
| "min": 0.2367386519908905, | |
| "max": 1.4733047485351562, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Entropy.sum": { | |
| "value": 7064.28125, | |
| "min": 7064.28125, | |
| "max": 44694.171875, | |
| "count": 33 | |
| }, | |
| "Pyramids.Step.mean": { | |
| "value": 989886.0, | |
| "min": 29952.0, | |
| "max": 989886.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Step.sum": { | |
| "value": 989886.0, | |
| "min": 29952.0, | |
| "max": 989886.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicValueEstimate.mean": { | |
| "value": 0.2921684682369232, | |
| "min": -0.11625519394874573, | |
| "max": 0.2921684682369232, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicValueEstimate.sum": { | |
| "value": 75.37946319580078, | |
| "min": -28.017501831054688, | |
| "max": 75.37946319580078, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndValueEstimate.mean": { | |
| "value": 0.016851382330060005, | |
| "min": -0.007969849742949009, | |
| "max": 0.40342140197753906, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndValueEstimate.sum": { | |
| "value": 4.347656726837158, | |
| "min": -1.992462396621704, | |
| "max": 95.61087036132812, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.PolicyLoss.mean": { | |
| "value": 0.0704993294253267, | |
| "min": 0.06596934255442441, | |
| "max": 0.0731857598556893, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.PolicyLoss.sum": { | |
| "value": 0.9869906119545739, | |
| "min": 0.48659106330971513, | |
| "max": 1.0870121684474368, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.ValueLoss.mean": { | |
| "value": 0.010154206059368554, | |
| "min": 0.0005932400581303815, | |
| "max": 0.011945566097676332, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.ValueLoss.sum": { | |
| "value": 0.14215888483115977, | |
| "min": 0.00830536081382534, | |
| "max": 0.16723792536746865, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.LearningRate.mean": { | |
| "value": 7.3097261348857125e-06, | |
| "min": 7.3097261348857125e-06, | |
| "max": 0.00029515063018788575, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.LearningRate.sum": { | |
| "value": 0.00010233616588839997, | |
| "min": 0.00010233616588839997, | |
| "max": 0.0035077958307347993, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Epsilon.mean": { | |
| "value": 0.10243654285714286, | |
| "min": 0.10243654285714286, | |
| "max": 0.19838354285714285, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Epsilon.sum": { | |
| "value": 1.4341116, | |
| "min": 1.3691136000000002, | |
| "max": 2.5692652000000002, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Beta.mean": { | |
| "value": 0.00025341063142857146, | |
| "min": 0.00025341063142857146, | |
| "max": 0.00983851593142857, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Beta.sum": { | |
| "value": 0.0035477488400000004, | |
| "min": 0.0035477488400000004, | |
| "max": 0.11694959348000002, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.RNDLoss.mean": { | |
| "value": 0.007888414897024632, | |
| "min": 0.007888414897024632, | |
| "max": 0.42303916811943054, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.RNDLoss.sum": { | |
| "value": 0.11043781042098999, | |
| "min": 0.11043781042098999, | |
| "max": 2.9612741470336914, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.EpisodeLength.mean": { | |
| "value": 587.3518518518518, | |
| "min": 534.4081632653061, | |
| "max": 999.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.EpisodeLength.sum": { | |
| "value": 31717.0, | |
| "min": 15984.0, | |
| "max": 32456.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.CumulativeReward.mean": { | |
| "value": 0.931033298097275, | |
| "min": -1.0000000521540642, | |
| "max": 1.1545654834601387, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.CumulativeReward.sum": { | |
| "value": 50.275798097252846, | |
| "min": -32.000001668930054, | |
| "max": 66.96479804068804, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicReward.mean": { | |
| "value": 0.931033298097275, | |
| "min": -1.0000000521540642, | |
| "max": 1.1545654834601387, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicReward.sum": { | |
| "value": 50.275798097252846, | |
| "min": -32.000001668930054, | |
| "max": 66.96479804068804, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndReward.mean": { | |
| "value": 0.047649473939263436, | |
| "min": 0.0440717160155804, | |
| "max": 8.777341455221176, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndReward.sum": { | |
| "value": 2.5730715927202255, | |
| "min": 2.4191946706123417, | |
| "max": 140.43746328353882, | |
| "count": 33 | |
| }, | |
| "Pyramids.IsTraining.mean": { | |
| "value": 1.0, | |
| "min": 1.0, | |
| "max": 1.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.IsTraining.sum": { | |
| "value": 1.0, | |
| "min": 1.0, | |
| "max": 1.0, | |
| "count": 33 | |
| } | |
| }, | |
| "metadata": { | |
| "timer_format_version": "0.1.0", | |
| "start_time_seconds": "1790945202", | |
| "python_version": "3.10.12 (main, Jul 5 2023, 18:54:27) [GCC 11.2.0]", | |
| "command_line_arguments": "/kaggle/working/mlagents-py310/bin/mlagents-learn /kaggle/working/ml-agents/config/ppo/PyramidsRND.yaml --env=/kaggle/working/ml-agents/ml-agents/training-envs-executables/linux/Pyramids/Pyramids --run-id=Pyramids Training 2 --no-graphics", | |
| "mlagents_version": "1.2.0.dev0", | |
| "mlagents_envs_version": "1.2.0.dev0", | |
| "communication_protocol_version": "1.5.0", | |
| "pytorch_version": "2.8.0+cu128", | |
| "numpy_version": "1.23.5", | |
| "end_time_seconds": "1790947155" | |
| }, | |
| "total": 1952.8264755500004, | |
| "count": 1, | |
| "self": 0.38176332900093257, | |
| "children": { | |
| "run_training.setup": { | |
| "total": 0.020500585999798204, | |
| "count": 1, | |
| "self": 0.020500585999798204 | |
| }, | |
| "TrainerController.start_learning": { | |
| "total": 1952.4242116349997, | |
| "count": 1, | |
| "self": 1.2750989819887764, | |
| "children": { | |
| "TrainerController._reset_env": { | |
| "total": 2.4783276440002737, | |
| "count": 1, | |
| "self": 2.4783276440002737 | |
| }, | |
| "TrainerController.advance": { | |
| "total": 1948.5892898170105, | |
| "count": 63435, | |
| "self": 1.3246139688817493, | |
| "children": { | |
| "env_step": { | |
| "total": 1334.8619289960707, | |
| "count": 63435, | |
| "self": 1185.678660379237, | |
| "children": { | |
| "SubprocessEnvManager._take_step": { | |
| "total": 148.40889680489636, | |
| "count": 63435, | |
| "self": 4.570508903868358, | |
| "children": { | |
| "TorchPolicy.evaluate": { | |
| "total": 143.838387901028, | |
| "count": 62561, | |
| "self": 143.838387901028 | |
| } | |
| } | |
| }, | |
| "workers": { | |
| "total": 0.7743718119372716, | |
| "count": 63435, | |
| "self": 0.0, | |
| "children": { | |
| "worker_root": { | |
| "total": 1949.4337058840592, | |
| "count": 63435, | |
| "is_parallel": true, | |
| "self": 859.9815858720522, | |
| "children": { | |
| "run_training.setup": { | |
| "total": 0.0, | |
| "count": 0, | |
| "is_parallel": true, | |
| "self": 0.0, | |
| "children": { | |
| "steps_from_proto": { | |
| "total": 0.002266974999656668, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0006747029992766329, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 0.001592272000380035, | |
| "count": 8, | |
| "is_parallel": true, | |
| "self": 0.001592272000380035 | |
| } | |
| } | |
| }, | |
| "UnityEnvironment.step": { | |
| "total": 0.04037781399983942, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.00035215799925936153, | |
| "children": { | |
| "UnityEnvironment._generate_step_input": { | |
| "total": 0.000614454999777081, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.000614454999777081 | |
| }, | |
| "communicator.exchange": { | |
| "total": 0.03827569900022354, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.03827569900022354 | |
| }, | |
| "steps_from_proto": { | |
| "total": 0.0011355020005794358, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0003705809995153686, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 0.0007649210010640672, | |
| "count": 8, | |
| "is_parallel": true, | |
| "self": 0.0007649210010640672 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "UnityEnvironment.step": { | |
| "total": 1089.452120012007, | |
| "count": 63434, | |
| "is_parallel": true, | |
| "self": 25.176381291842517, | |
| "children": { | |
| "UnityEnvironment._generate_step_input": { | |
| "total": 18.16159100916957, | |
| "count": 63434, | |
| "is_parallel": true, | |
| "self": 18.16159100916957 | |
| }, | |
| "communicator.exchange": { | |
| "total": 971.828856577954, | |
| "count": 63434, | |
| "is_parallel": true, | |
| "self": 971.828856577954 | |
| }, | |
| "steps_from_proto": { | |
| "total": 74.28529113304103, | |
| "count": 63434, | |
| "is_parallel": true, | |
| "self": 15.12161121693498, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 59.16367991610605, | |
| "count": 507472, | |
| "is_parallel": true, | |
| "self": 59.16367991610605 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "trainer_advance": { | |
| "total": 612.402746852058, | |
| "count": 63435, | |
| "self": 2.4562060362122793, | |
| "children": { | |
| "process_trajectory": { | |
| "total": 107.42076671183713, | |
| "count": 63435, | |
| "self": 107.22599126383648, | |
| "children": { | |
| "RLTrainer._checkpoint": { | |
| "total": 0.19477544800065516, | |
| "count": 2, | |
| "self": 0.19477544800065516 | |
| } | |
| } | |
| }, | |
| "_update_policy": { | |
| "total": 502.52577410400863, | |
| "count": 448, | |
| "self": 261.3169638819718, | |
| "children": { | |
| "TorchPPOOptimizer.update": { | |
| "total": 241.20881022203685, | |
| "count": 22815, | |
| "self": 241.20881022203685 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "trainer_threads": { | |
| "total": 1.1290003385511227e-06, | |
| "count": 1, | |
| "self": 1.1290003385511227e-06 | |
| }, | |
| "TrainerController._save_models": { | |
| "total": 0.0814940629998091, | |
| "count": 1, | |
| "self": 0.0008690370004842407, | |
| "children": { | |
| "RLTrainer._checkpoint": { | |
| "total": 0.08062502599932486, | |
| "count": 1, | |
| "self": 0.08062502599932486 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } |