Reinforcement Learning
ml-agents
TensorBoard
ONNX
Pyramids
deep-reinforcement-learning
ML-Agents-Pyramids
Instructions to use leuyendt/ppo-Pyramids with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- ml-agents
How to use leuyendt/ppo-Pyramids with ml-agents:
mlagents-load-from-hf --repo-id="leuyendt/ppo-Pyramids" --local-dir="./download: string[]s"
- Notebooks
- Google Colab
- Kaggle
| { | |
| "name": "root", | |
| "gauges": { | |
| "Pyramids.Policy.Entropy.mean": { | |
| "value": 0.35594111680984497, | |
| "min": 0.3461613953113556, | |
| "max": 1.4570527076721191, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Entropy.sum": { | |
| "value": 10638.3681640625, | |
| "min": 10384.841796875, | |
| "max": 44201.15234375, | |
| "count": 33 | |
| }, | |
| "Pyramids.Step.mean": { | |
| "value": 989873.0, | |
| "min": 29952.0, | |
| "max": 989873.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Step.sum": { | |
| "value": 989873.0, | |
| "min": 29952.0, | |
| "max": 989873.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicValueEstimate.mean": { | |
| "value": 0.3650670647621155, | |
| "min": -0.11311739683151245, | |
| "max": 0.4035182297229767, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicValueEstimate.sum": { | |
| "value": 96.01264190673828, | |
| "min": -26.808822631835938, | |
| "max": 107.33584594726562, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndValueEstimate.mean": { | |
| "value": 0.018011484295129776, | |
| "min": 0.0029153083451092243, | |
| "max": 0.3900042772293091, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndValueEstimate.sum": { | |
| "value": 4.737020492553711, | |
| "min": 0.7608954906463623, | |
| "max": 92.43101501464844, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.PolicyLoss.mean": { | |
| "value": 0.06970075355535041, | |
| "min": 0.0633668324061691, | |
| "max": 0.07355648497069327, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.PolicyLoss.sum": { | |
| "value": 0.9758105497749058, | |
| "min": 0.50149182003057, | |
| "max": 1.0884587479197174, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.ValueLoss.mean": { | |
| "value": 0.009763881259069811, | |
| "min": 0.00017259593206393626, | |
| "max": 0.012486893873998367, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.ValueLoss.sum": { | |
| "value": 0.13669433762697736, | |
| "min": 0.0024163430488951076, | |
| "max": 0.17931955715746378, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.LearningRate.mean": { | |
| "value": 7.595347468249998e-06, | |
| "min": 7.595347468249998e-06, | |
| "max": 0.00029515063018788575, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.LearningRate.sum": { | |
| "value": 0.00010633486455549997, | |
| "min": 0.00010633486455549997, | |
| "max": 0.0036352414882529, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Epsilon.mean": { | |
| "value": 0.10253175000000002, | |
| "min": 0.10253175000000002, | |
| "max": 0.19838354285714285, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Epsilon.sum": { | |
| "value": 1.4354445000000002, | |
| "min": 1.3886848, | |
| "max": 2.6173358, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Beta.mean": { | |
| "value": 0.00026292182499999995, | |
| "min": 0.00026292182499999995, | |
| "max": 0.00983851593142857, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Beta.sum": { | |
| "value": 0.0036809055499999995, | |
| "min": 0.0036809055499999995, | |
| "max": 0.12119353529, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.RNDLoss.mean": { | |
| "value": 0.010120609775185585, | |
| "min": 0.010120609775185585, | |
| "max": 0.41265568137168884, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.RNDLoss.sum": { | |
| "value": 0.1416885405778885, | |
| "min": 0.1416885405778885, | |
| "max": 2.888589859008789, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.EpisodeLength.mean": { | |
| "value": 506.7704918032787, | |
| "min": 450.9076923076923, | |
| "max": 999.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.EpisodeLength.sum": { | |
| "value": 30913.0, | |
| "min": 15984.0, | |
| "max": 33403.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.CumulativeReward.mean": { | |
| "value": 1.2308360434702186, | |
| "min": -1.0000000521540642, | |
| "max": 1.3647014658842513, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.CumulativeReward.sum": { | |
| "value": 75.08099865168333, | |
| "min": -31.98920165002346, | |
| "max": 91.43499821424484, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicReward.mean": { | |
| "value": 1.2308360434702186, | |
| "min": -1.0000000521540642, | |
| "max": 1.3647014658842513, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicReward.sum": { | |
| "value": 75.08099865168333, | |
| "min": -31.98920165002346, | |
| "max": 91.43499821424484, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndReward.mean": { | |
| "value": 0.05296870745187931, | |
| "min": 0.05017191919744877, | |
| "max": 8.304333683103323, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndReward.sum": { | |
| "value": 3.2310911545646377, | |
| "min": 3.07193310925868, | |
| "max": 132.86933892965317, | |
| "count": 33 | |
| }, | |
| "Pyramids.IsTraining.mean": { | |
| "value": 1.0, | |
| "min": 1.0, | |
| "max": 1.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.IsTraining.sum": { | |
| "value": 1.0, | |
| "min": 1.0, | |
| "max": 1.0, | |
| "count": 33 | |
| } | |
| }, | |
| "metadata": { | |
| "timer_format_version": "0.1.0", | |
| "start_time_seconds": "1781739437", | |
| "python_version": "3.10.12 (main, Jul 5 2023, 18:54:27) [GCC 11.2.0]", | |
| "command_line_arguments": "/usr/local/envs/mlagents_env/bin/mlagents-learn ./config/ppo/PyramidsRND.yaml --env=./training-envs-executables/linux/Pyramids/Pyramids --run-id=Pyramids Training --no-graphics --force", | |
| "mlagents_version": "1.2.0.dev0", | |
| "mlagents_envs_version": "1.2.0.dev0", | |
| "communication_protocol_version": "1.5.0", | |
| "pytorch_version": "2.8.0+cu128", | |
| "numpy_version": "1.23.5", | |
| "end_time_seconds": "1781742516" | |
| }, | |
| "total": 3078.9617160239995, | |
| "count": 1, | |
| "self": 0.8006007809981384, | |
| "children": { | |
| "run_training.setup": { | |
| "total": 0.02639121300126135, | |
| "count": 1, | |
| "self": 0.02639121300126135 | |
| }, | |
| "TrainerController.start_learning": { | |
| "total": 3078.13472403, | |
| "count": 1, | |
| "self": 2.021102609258378, | |
| "children": { | |
| "TrainerController._reset_env": { | |
| "total": 3.1343409159999283, | |
| "count": 1, | |
| "self": 3.1343409159999283 | |
| }, | |
| "TrainerController.advance": { | |
| "total": 3072.895861940742, | |
| "count": 63681, | |
| "self": 2.2610821235630283, | |
| "children": { | |
| "env_step": { | |
| "total": 2119.4864100261366, | |
| "count": 63681, | |
| "self": 1973.1252936381334, | |
| "children": { | |
| "SubprocessEnvManager._take_step": { | |
| "total": 145.0368187609838, | |
| "count": 63681, | |
| "self": 6.22597357833547, | |
| "children": { | |
| "TorchPolicy.evaluate": { | |
| "total": 138.81084518264834, | |
| "count": 62567, | |
| "self": 138.81084518264834 | |
| } | |
| } | |
| }, | |
| "workers": { | |
| "total": 1.324297627019405, | |
| "count": 63681, | |
| "self": 0.0, | |
| "children": { | |
| "worker_root": { | |
| "total": 3070.886805566108, | |
| "count": 63681, | |
| "is_parallel": true, | |
| "self": 1266.7450732819343, | |
| "children": { | |
| "run_training.setup": { | |
| "total": 0.0, | |
| "count": 0, | |
| "is_parallel": true, | |
| "self": 0.0, | |
| "children": { | |
| "steps_from_proto": { | |
| "total": 0.005319207999491482, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0038990379980532452, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 0.0014201700014382368, | |
| "count": 8, | |
| "is_parallel": true, | |
| "self": 0.0014201700014382368 | |
| } | |
| } | |
| }, | |
| "UnityEnvironment.step": { | |
| "total": 0.0814024569990579, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0006746790004399372, | |
| "children": { | |
| "UnityEnvironment._generate_step_input": { | |
| "total": 0.0005536459993891185, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0005536459993891185 | |
| }, | |
| "communicator.exchange": { | |
| "total": 0.07828938099919469, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.07828938099919469 | |
| }, | |
| "steps_from_proto": { | |
| "total": 0.0018847510000341572, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.00039818599907448515, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 0.001486565000959672, | |
| "count": 8, | |
| "is_parallel": true, | |
| "self": 0.001486565000959672 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "UnityEnvironment.step": { | |
| "total": 1804.1417322841735, | |
| "count": 63680, | |
| "is_parallel": true, | |
| "self": 46.25935951645624, | |
| "children": { | |
| "UnityEnvironment._generate_step_input": { | |
| "total": 29.797233976740245, | |
| "count": 63680, | |
| "is_parallel": true, | |
| "self": 29.797233976740245 | |
| }, | |
| "communicator.exchange": { | |
| "total": 1590.4287348810776, | |
| "count": 63680, | |
| "is_parallel": true, | |
| "self": 1590.4287348810776 | |
| }, | |
| "steps_from_proto": { | |
| "total": 137.65640390989938, | |
| "count": 63680, | |
| "is_parallel": true, | |
| "self": 27.195411567856354, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 110.46099234204303, | |
| "count": 509440, | |
| "is_parallel": true, | |
| "self": 110.46099234204303 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "trainer_advance": { | |
| "total": 951.1483697910426, | |
| "count": 63681, | |
| "self": 4.07464904895096, | |
| "children": { | |
| "process_trajectory": { | |
| "total": 136.31217278908298, | |
| "count": 63681, | |
| "self": 135.87137759108555, | |
| "children": { | |
| "RLTrainer._checkpoint": { | |
| "total": 0.4407951979974314, | |
| "count": 2, | |
| "self": 0.4407951979974314 | |
| } | |
| } | |
| }, | |
| "_update_policy": { | |
| "total": 810.7615479530086, | |
| "count": 456, | |
| "self": 316.63814828598515, | |
| "children": { | |
| "TorchPPOOptimizer.update": { | |
| "total": 494.1233996670235, | |
| "count": 22806, | |
| "self": 494.1233996670235 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "trainer_threads": { | |
| "total": 1.0370004019932821e-06, | |
| "count": 1, | |
| "self": 1.0370004019932821e-06 | |
| }, | |
| "TrainerController._save_models": { | |
| "total": 0.08341752699925564, | |
| "count": 1, | |
| "self": 0.0018388560001767473, | |
| "children": { | |
| "RLTrainer._checkpoint": { | |
| "total": 0.0815786709990789, | |
| "count": 1, | |
| "self": 0.0815786709990789 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } |