Reinforcement Learning
ml-agents
TensorBoard
ONNX
Pyramids
deep-reinforcement-learning
ML-Agents-Pyramids
Instructions to use liamleirs/ppo-PyramidsRND with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- ml-agents
How to use liamleirs/ppo-PyramidsRND with ml-agents:
mlagents-load-from-hf --repo-id="liamleirs/ppo-PyramidsRND" --local-dir="./download: string[]s"
- Notebooks
- Google Colab
- Kaggle
| { | |
| "name": "root", | |
| "gauges": { | |
| "Pyramids.Policy.Entropy.mean": { | |
| "value": 0.28874826431274414, | |
| "min": 0.28799375891685486, | |
| "max": 1.4167274236679077, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Entropy.sum": { | |
| "value": 8704.02734375, | |
| "min": 8612.1650390625, | |
| "max": 42977.84375, | |
| "count": 33 | |
| }, | |
| "Pyramids.Step.mean": { | |
| "value": 989934.0, | |
| "min": 29952.0, | |
| "max": 989934.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Step.sum": { | |
| "value": 989934.0, | |
| "min": 29952.0, | |
| "max": 989934.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicValueEstimate.mean": { | |
| "value": 0.5555307865142822, | |
| "min": -0.09557277709245682, | |
| "max": 0.6632574200630188, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicValueEstimate.sum": { | |
| "value": 151.659912109375, | |
| "min": -22.93746566772461, | |
| "max": 191.681396484375, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndValueEstimate.mean": { | |
| "value": 0.00242367060855031, | |
| "min": -0.1786683201789856, | |
| "max": 0.3713178038597107, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndValueEstimate.sum": { | |
| "value": 0.6616621017456055, | |
| "min": -46.989768981933594, | |
| "max": 88.0023193359375, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.PolicyLoss.mean": { | |
| "value": 0.07172153633984549, | |
| "min": 0.06530367043322056, | |
| "max": 0.07490418907920164, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.PolicyLoss.sum": { | |
| "value": 1.0041015087578369, | |
| "min": 0.4977167543397156, | |
| "max": 1.079499623665755, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.ValueLoss.mean": { | |
| "value": 0.014853488293565097, | |
| "min": 0.00026929999669677445, | |
| "max": 0.018629020147186346, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.ValueLoss.sum": { | |
| "value": 0.20794883610991136, | |
| "min": 0.0032315999603612935, | |
| "max": 0.26080628206060885, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.LearningRate.mean": { | |
| "value": 7.384468967114283e-06, | |
| "min": 7.384468967114283e-06, | |
| "max": 0.00029515063018788575, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.LearningRate.sum": { | |
| "value": 0.00010338256553959997, | |
| "min": 0.00010338256553959997, | |
| "max": 0.0036345421884860004, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Epsilon.mean": { | |
| "value": 0.10246145714285713, | |
| "min": 0.10246145714285713, | |
| "max": 0.19838354285714285, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Epsilon.sum": { | |
| "value": 1.4344603999999999, | |
| "min": 1.3886848, | |
| "max": 2.662620300000001, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Beta.mean": { | |
| "value": 0.00025589956857142857, | |
| "min": 0.00025589956857142857, | |
| "max": 0.00983851593142857, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Beta.sum": { | |
| "value": 0.00358259396, | |
| "min": 0.00358259396, | |
| "max": 0.12117024859999999, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.RNDLoss.mean": { | |
| "value": 0.010907362215220928, | |
| "min": 0.010907362215220928, | |
| "max": 0.46725723147392273, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.RNDLoss.sum": { | |
| "value": 0.15270307660102844, | |
| "min": 0.15270307660102844, | |
| "max": 3.2708005905151367, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.EpisodeLength.mean": { | |
| "value": 326.01162790697674, | |
| "min": 286.3592233009709, | |
| "max": 999.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.EpisodeLength.sum": { | |
| "value": 28037.0, | |
| "min": 15984.0, | |
| "max": 32198.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.CumulativeReward.mean": { | |
| "value": 1.6068045695622761, | |
| "min": -1.0000000521540642, | |
| "max": 1.694215516731577, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.CumulativeReward.sum": { | |
| "value": 139.79199755191803, | |
| "min": -29.50300160050392, | |
| "max": 174.50419822335243, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicReward.mean": { | |
| "value": 1.6068045695622761, | |
| "min": -1.0000000521540642, | |
| "max": 1.694215516731577, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicReward.sum": { | |
| "value": 139.79199755191803, | |
| "min": -29.50300160050392, | |
| "max": 174.50419822335243, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndReward.mean": { | |
| "value": 0.036753509975615196, | |
| "min": 0.034706200959016255, | |
| "max": 8.72353034093976, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndReward.sum": { | |
| "value": 3.197555367878522, | |
| "min": 3.197555367878522, | |
| "max": 139.57648545503616, | |
| "count": 33 | |
| }, | |
| "Pyramids.IsTraining.mean": { | |
| "value": 1.0, | |
| "min": 1.0, | |
| "max": 1.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.IsTraining.sum": { | |
| "value": 1.0, | |
| "min": 1.0, | |
| "max": 1.0, | |
| "count": 33 | |
| } | |
| }, | |
| "metadata": { | |
| "timer_format_version": "0.1.0", | |
| "start_time_seconds": "1785244211", | |
| "python_version": "3.10.12 (main, Jul 5 2023, 18:54:27) [GCC 11.2.0]", | |
| "command_line_arguments": "/usr/local/bin/mlagents-learn ../config/ppo/PyramidsRND.yaml --env=./training-envs-executables/linux/Pyramids/Pyramids --run-id=Pyramids Training --no-graphics", | |
| "mlagents_version": "1.2.0.dev0", | |
| "mlagents_envs_version": "1.2.0.dev0", | |
| "communication_protocol_version": "1.5.0", | |
| "pytorch_version": "2.8.0+cu128", | |
| "numpy_version": "1.23.5", | |
| "end_time_seconds": "1785246685" | |
| }, | |
| "total": 2474.704850659, | |
| "count": 1, | |
| "self": 0.48188655100057076, | |
| "children": { | |
| "run_training.setup": { | |
| "total": 0.02407745599998634, | |
| "count": 1, | |
| "self": 0.02407745599998634 | |
| }, | |
| "TrainerController.start_learning": { | |
| "total": 2474.1988866519996, | |
| "count": 1, | |
| "self": 1.4009894140590404, | |
| "children": { | |
| "TrainerController._reset_env": { | |
| "total": 2.1001859259999947, | |
| "count": 1, | |
| "self": 2.1001859259999947 | |
| }, | |
| "TrainerController.advance": { | |
| "total": 2470.6249303529407, | |
| "count": 64223, | |
| "self": 1.424324903933666, | |
| "children": { | |
| "env_step": { | |
| "total": 1840.6942319810125, | |
| "count": 64223, | |
| "self": 1688.9804382189952, | |
| "children": { | |
| "SubprocessEnvManager._take_step": { | |
| "total": 150.87593891800225, | |
| "count": 64223, | |
| "self": 4.6893423479334615, | |
| "children": { | |
| "TorchPolicy.evaluate": { | |
| "total": 146.1865965700688, | |
| "count": 62571, | |
| "self": 146.1865965700688 | |
| } | |
| } | |
| }, | |
| "workers": { | |
| "total": 0.8378548440150553, | |
| "count": 64223, | |
| "self": 0.0, | |
| "children": { | |
| "worker_root": { | |
| "total": 2468.3344161458667, | |
| "count": 64223, | |
| "is_parallel": true, | |
| "self": 898.932100925857, | |
| "children": { | |
| "run_training.setup": { | |
| "total": 0.0, | |
| "count": 0, | |
| "is_parallel": true, | |
| "self": 0.0, | |
| "children": { | |
| "steps_from_proto": { | |
| "total": 0.0017611549999401177, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.000586303000090993, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 0.0011748519998491247, | |
| "count": 8, | |
| "is_parallel": true, | |
| "self": 0.0011748519998491247 | |
| } | |
| } | |
| }, | |
| "UnityEnvironment.step": { | |
| "total": 0.05134154400002444, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0005398519999744167, | |
| "children": { | |
| "UnityEnvironment._generate_step_input": { | |
| "total": 0.0004943330000060087, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0004943330000060087 | |
| }, | |
| "communicator.exchange": { | |
| "total": 0.04867942300006689, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.04867942300006689 | |
| }, | |
| "steps_from_proto": { | |
| "total": 0.001627935999977126, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0003827469997759181, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 0.0012451890002012078, | |
| "count": 8, | |
| "is_parallel": true, | |
| "self": 0.0012451890002012078 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "UnityEnvironment.step": { | |
| "total": 1569.4023152200098, | |
| "count": 64222, | |
| "is_parallel": true, | |
| "self": 33.826823968032386, | |
| "children": { | |
| "UnityEnvironment._generate_step_input": { | |
| "total": 23.486220446953894, | |
| "count": 64222, | |
| "is_parallel": true, | |
| "self": 23.486220446953894 | |
| }, | |
| "communicator.exchange": { | |
| "total": 1402.7791468410726, | |
| "count": 64222, | |
| "is_parallel": true, | |
| "self": 1402.7791468410726 | |
| }, | |
| "steps_from_proto": { | |
| "total": 109.3101239639509, | |
| "count": 64222, | |
| "is_parallel": true, | |
| "self": 22.631807794143924, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 86.67831616980698, | |
| "count": 513776, | |
| "is_parallel": true, | |
| "self": 86.67831616980698 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "trainer_advance": { | |
| "total": 628.5063734679945, | |
| "count": 64223, | |
| "self": 2.7720307919912557, | |
| "children": { | |
| "process_trajectory": { | |
| "total": 113.23971866400188, | |
| "count": 64223, | |
| "self": 113.06028520100153, | |
| "children": { | |
| "RLTrainer._checkpoint": { | |
| "total": 0.17943346300035046, | |
| "count": 2, | |
| "self": 0.17943346300035046 | |
| } | |
| } | |
| }, | |
| "_update_policy": { | |
| "total": 512.4946240120014, | |
| "count": 456, | |
| "self": 272.69853448904064, | |
| "children": { | |
| "TorchPPOOptimizer.update": { | |
| "total": 239.79608952296076, | |
| "count": 22791, | |
| "self": 239.79608952296076 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "trainer_threads": { | |
| "total": 8.580000212532468e-07, | |
| "count": 1, | |
| "self": 8.580000212532468e-07 | |
| }, | |
| "TrainerController._save_models": { | |
| "total": 0.07278010099980747, | |
| "count": 1, | |
| "self": 0.0010728240004027612, | |
| "children": { | |
| "RLTrainer._checkpoint": { | |
| "total": 0.07170727699940471, | |
| "count": 1, | |
| "self": 0.07170727699940471 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } |