Reinforcement Learning
ml-agents
TensorBoard
ONNX
Pyramids
deep-reinforcement-learning
ML-Agents-Pyramids
Instructions to use MathieuGALINIER/ppo-Pyramids with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- ml-agents
How to use MathieuGALINIER/ppo-Pyramids with ml-agents:
mlagents-load-from-hf --repo-id="MathieuGALINIER/ppo-Pyramids" --local-dir="./download: string[]s"
- Notebooks
- Google Colab
- Kaggle
| { | |
| "name": "root", | |
| "gauges": { | |
| "Pyramids.Policy.Entropy.mean": { | |
| "value": 0.44912347197532654, | |
| "min": 0.44829437136650085, | |
| "max": 1.459127426147461, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Entropy.sum": { | |
| "value": 13358.728515625, | |
| "min": 13358.728515625, | |
| "max": 44264.08984375, | |
| "count": 33 | |
| }, | |
| "Pyramids.Step.mean": { | |
| "value": 989897.0, | |
| "min": 29952.0, | |
| "max": 989897.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Step.sum": { | |
| "value": 989897.0, | |
| "min": 29952.0, | |
| "max": 989897.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicValueEstimate.mean": { | |
| "value": 0.32942643761634827, | |
| "min": -0.12526600062847137, | |
| "max": 0.3584676682949066, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicValueEstimate.sum": { | |
| "value": 84.66259765625, | |
| "min": -30.18910789489746, | |
| "max": 92.84312438964844, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndValueEstimate.mean": { | |
| "value": 0.004904638044536114, | |
| "min": 0.00212547997944057, | |
| "max": 0.22454407811164856, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndValueEstimate.sum": { | |
| "value": 1.260491967201233, | |
| "min": 0.5377464294433594, | |
| "max": 54.1151237487793, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.PolicyLoss.mean": { | |
| "value": 0.06985345459681759, | |
| "min": 0.0657316405434228, | |
| "max": 0.07379627501598648, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.PolicyLoss.sum": { | |
| "value": 0.9779483643554462, | |
| "min": 0.49256165636651184, | |
| "max": 1.0397710791245722, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.ValueLoss.mean": { | |
| "value": 0.011089424375920013, | |
| "min": 0.0004076620917615216, | |
| "max": 0.012793812375180314, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.ValueLoss.sum": { | |
| "value": 0.15525194126288017, | |
| "min": 0.005299607192899781, | |
| "max": 0.17911337325252438, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.LearningRate.mean": { | |
| "value": 7.597104610521427e-06, | |
| "min": 7.597104610521427e-06, | |
| "max": 0.00029515063018788575, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.LearningRate.sum": { | |
| "value": 0.00010635946454729998, | |
| "min": 0.00010635946454729998, | |
| "max": 0.0032561567146144996, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Epsilon.mean": { | |
| "value": 0.10253233571428573, | |
| "min": 0.10253233571428573, | |
| "max": 0.19838354285714285, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Epsilon.sum": { | |
| "value": 1.4354527000000001, | |
| "min": 1.3886848, | |
| "max": 2.3853855000000004, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Beta.mean": { | |
| "value": 0.0002629803378571429, | |
| "min": 0.0002629803378571429, | |
| "max": 0.00983851593142857, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Beta.sum": { | |
| "value": 0.00368172473, | |
| "min": 0.00368172473, | |
| "max": 0.10856001145000001, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.RNDLoss.mean": { | |
| "value": 0.01090999972075224, | |
| "min": 0.010771428234875202, | |
| "max": 0.43669724464416504, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.RNDLoss.sum": { | |
| "value": 0.1527400016784668, | |
| "min": 0.15079998970031738, | |
| "max": 3.0568807125091553, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.EpisodeLength.mean": { | |
| "value": 518.578947368421, | |
| "min": 490.78333333333336, | |
| "max": 999.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.EpisodeLength.sum": { | |
| "value": 29559.0, | |
| "min": 15984.0, | |
| "max": 32223.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.CumulativeReward.mean": { | |
| "value": 1.1304210241426502, | |
| "min": -1.0000000521540642, | |
| "max": 1.2424666322767735, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.CumulativeReward.sum": { | |
| "value": 64.43399837613106, | |
| "min": -29.96380166709423, | |
| "max": 74.54799793660641, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicReward.mean": { | |
| "value": 1.1304210241426502, | |
| "min": -1.0000000521540642, | |
| "max": 1.2424666322767735, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicReward.sum": { | |
| "value": 64.43399837613106, | |
| "min": -29.96380166709423, | |
| "max": 74.54799793660641, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndReward.mean": { | |
| "value": 0.0587151825526043, | |
| "min": 0.054262425736912215, | |
| "max": 8.471816313453019, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndReward.sum": { | |
| "value": 3.346765405498445, | |
| "min": 3.255745544214733, | |
| "max": 135.5490610152483, | |
| "count": 33 | |
| }, | |
| "Pyramids.IsTraining.mean": { | |
| "value": 1.0, | |
| "min": 1.0, | |
| "max": 1.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.IsTraining.sum": { | |
| "value": 1.0, | |
| "min": 1.0, | |
| "max": 1.0, | |
| "count": 33 | |
| } | |
| }, | |
| "metadata": { | |
| "timer_format_version": "0.1.0", | |
| "start_time_seconds": "1787659350", | |
| "python_version": "3.10.12 (main, Jul 5 2023, 18:54:27) [GCC 11.2.0]", | |
| "command_line_arguments": "/usr/local/bin/mlagents-learn ./config/ppo/PyramidsRND.yaml --env=./training-envs-executables/linux/Pyramids/Pyramids --run-id=Pyramids Training --no-graphics", | |
| "mlagents_version": "1.2.0.dev0", | |
| "mlagents_envs_version": "1.2.0.dev0", | |
| "communication_protocol_version": "1.5.0", | |
| "pytorch_version": "2.8.0+cu128", | |
| "numpy_version": "1.23.5", | |
| "end_time_seconds": "1787661855" | |
| }, | |
| "total": 2505.5211165539995, | |
| "count": 1, | |
| "self": 0.5296329679995324, | |
| "children": { | |
| "run_training.setup": { | |
| "total": 0.02452394700003424, | |
| "count": 1, | |
| "self": 0.02452394700003424 | |
| }, | |
| "TrainerController.start_learning": { | |
| "total": 2504.966959639, | |
| "count": 1, | |
| "self": 1.6317040978974546, | |
| "children": { | |
| "TrainerController._reset_env": { | |
| "total": 2.1734966310000345, | |
| "count": 1, | |
| "self": 2.1734966310000345 | |
| }, | |
| "TrainerController.advance": { | |
| "total": 2501.0694717901015, | |
| "count": 63380, | |
| "self": 1.5876212550460878, | |
| "children": { | |
| "env_step": { | |
| "total": 1828.9176462840078, | |
| "count": 63380, | |
| "self": 1656.505584552021, | |
| "children": { | |
| "SubprocessEnvManager._take_step": { | |
| "total": 171.47006853800804, | |
| "count": 63380, | |
| "self": 5.202593275959316, | |
| "children": { | |
| "TorchPolicy.evaluate": { | |
| "total": 166.26747526204872, | |
| "count": 62564, | |
| "self": 166.26747526204872 | |
| } | |
| } | |
| }, | |
| "workers": { | |
| "total": 0.9419931939787602, | |
| "count": 63380, | |
| "self": 0.0, | |
| "children": { | |
| "worker_root": { | |
| "total": 2498.869685851089, | |
| "count": 63380, | |
| "is_parallel": true, | |
| "self": 975.2516152960975, | |
| "children": { | |
| "run_training.setup": { | |
| "total": 0.0, | |
| "count": 0, | |
| "is_parallel": true, | |
| "self": 0.0, | |
| "children": { | |
| "steps_from_proto": { | |
| "total": 0.0018159820001528715, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0005663460001414933, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 0.0012496360000113782, | |
| "count": 8, | |
| "is_parallel": true, | |
| "self": 0.0012496360000113782 | |
| } | |
| } | |
| }, | |
| "UnityEnvironment.step": { | |
| "total": 0.0623888749996695, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0005418359992290789, | |
| "children": { | |
| "UnityEnvironment._generate_step_input": { | |
| "total": 0.0005267400001685019, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0005267400001685019 | |
| }, | |
| "communicator.exchange": { | |
| "total": 0.05959522500006642, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.05959522500006642 | |
| }, | |
| "steps_from_proto": { | |
| "total": 0.0017250740002054954, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.00038487099982376094, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 0.0013402030003817345, | |
| "count": 8, | |
| "is_parallel": true, | |
| "self": 0.0013402030003817345 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "UnityEnvironment.step": { | |
| "total": 1523.6180705549914, | |
| "count": 63379, | |
| "is_parallel": true, | |
| "self": 37.62466095101718, | |
| "children": { | |
| "UnityEnvironment._generate_step_input": { | |
| "total": 26.640666370962208, | |
| "count": 63379, | |
| "is_parallel": true, | |
| "self": 26.640666370962208 | |
| }, | |
| "communicator.exchange": { | |
| "total": 1334.8779385270013, | |
| "count": 63379, | |
| "is_parallel": true, | |
| "self": 1334.8779385270013 | |
| }, | |
| "steps_from_proto": { | |
| "total": 124.47480470601067, | |
| "count": 63379, | |
| "is_parallel": true, | |
| "self": 25.655858211909617, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 98.81894649410106, | |
| "count": 507032, | |
| "is_parallel": true, | |
| "self": 98.81894649410106 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "trainer_advance": { | |
| "total": 670.5642042510476, | |
| "count": 63380, | |
| "self": 2.7705597691669936, | |
| "children": { | |
| "process_trajectory": { | |
| "total": 116.74718594687738, | |
| "count": 63380, | |
| "self": 116.52374694987702, | |
| "children": { | |
| "RLTrainer._checkpoint": { | |
| "total": 0.22343899700035763, | |
| "count": 2, | |
| "self": 0.22343899700035763 | |
| } | |
| } | |
| }, | |
| "_update_policy": { | |
| "total": 551.0464585350032, | |
| "count": 442, | |
| "self": 294.9093710510001, | |
| "children": { | |
| "TorchPPOOptimizer.update": { | |
| "total": 256.13708748400313, | |
| "count": 22800, | |
| "self": 256.13708748400313 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "trainer_threads": { | |
| "total": 1.0550002116360702e-06, | |
| "count": 1, | |
| "self": 1.0550002116360702e-06 | |
| }, | |
| "TrainerController._save_models": { | |
| "total": 0.0922860650007351, | |
| "count": 1, | |
| "self": 0.001150422001046536, | |
| "children": { | |
| "RLTrainer._checkpoint": { | |
| "total": 0.09113564299968857, | |
| "count": 1, | |
| "self": 0.09113564299968857 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } |