Reinforcement Learning
ml-agents
TensorBoard
ONNX
Pyramids
deep-reinforcement-learning
ML-Agents-Pyramids
Instructions to use an0maliee/ppo-Pyramid with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- ml-agents
How to use an0maliee/ppo-Pyramid with ml-agents:
mlagents-load-from-hf --repo-id="an0maliee/ppo-Pyramid" --local-dir="./download: string[]s"
- Notebooks
- Google Colab
- Kaggle
| { | |
| "name": "root", | |
| "gauges": { | |
| "Pyramids.Policy.Entropy.mean": { | |
| "value": 0.37783801555633545, | |
| "min": 0.3408486545085907, | |
| "max": 1.3605443239212036, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Entropy.sum": { | |
| "value": 11407.685546875, | |
| "min": 10209.0986328125, | |
| "max": 32914.2890625, | |
| "count": 33 | |
| }, | |
| "Pyramids.Step.mean": { | |
| "value": 989918.0, | |
| "min": 29951.0, | |
| "max": 989918.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Step.sum": { | |
| "value": 989918.0, | |
| "min": 29951.0, | |
| "max": 989918.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicValueEstimate.mean": { | |
| "value": 0.2156532108783722, | |
| "min": -0.17285645008087158, | |
| "max": 0.28188860416412354, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicValueEstimate.sum": { | |
| "value": 54.56026077270508, | |
| "min": -32.66986846923828, | |
| "max": 73.00914764404297, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndValueEstimate.mean": { | |
| "value": 0.03783803433179855, | |
| "min": 0.0030573680996894836, | |
| "max": 0.472737580537796, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndValueEstimate.sum": { | |
| "value": 9.573022842407227, | |
| "min": 0.7735141515731812, | |
| "max": 89.45181274414062, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.PolicyLoss.mean": { | |
| "value": 0.06721809591648967, | |
| "min": 0.06562428259092164, | |
| "max": 0.07318220606786134, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.PolicyLoss.sum": { | |
| "value": 0.9410533428308553, | |
| "min": 0.40870321522695424, | |
| "max": 1.062153641302846, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.ValueLoss.mean": { | |
| "value": 0.008414831094601791, | |
| "min": 0.000695455142374161, | |
| "max": 0.015856676001355433, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.ValueLoss.sum": { | |
| "value": 0.11780763532442506, | |
| "min": 0.009040916850864092, | |
| "max": 0.1697651332591811, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.LearningRate.mean": { | |
| "value": 7.317204703821425e-06, | |
| "min": 7.317204703821425e-06, | |
| "max": 0.0002942925519024833, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.LearningRate.sum": { | |
| "value": 0.00010244086585349996, | |
| "min": 0.00010244086585349996, | |
| "max": 0.0034910743363085993, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Epsilon.mean": { | |
| "value": 0.10243903571428573, | |
| "min": 0.10243903571428573, | |
| "max": 0.19809751666666667, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Epsilon.sum": { | |
| "value": 1.4341465000000002, | |
| "min": 1.1885851, | |
| "max": 2.4852986, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Beta.mean": { | |
| "value": 0.00025365966785714274, | |
| "min": 0.00025365966785714274, | |
| "max": 0.009809941915, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Beta.sum": { | |
| "value": 0.0035512353499999984, | |
| "min": 0.0035512353499999984, | |
| "max": 0.11638277086, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.RNDLoss.mean": { | |
| "value": 0.027771679684519768, | |
| "min": 0.027766002342104912, | |
| "max": 0.5217909216880798, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.RNDLoss.sum": { | |
| "value": 0.38880351185798645, | |
| "min": 0.38872402906417847, | |
| "max": 3.1307456493377686, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.EpisodeLength.mean": { | |
| "value": 629.4468085106383, | |
| "min": 529.6071428571429, | |
| "max": 998.9375, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.EpisodeLength.sum": { | |
| "value": 29584.0, | |
| "min": 15983.0, | |
| "max": 32644.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.CumulativeReward.mean": { | |
| "value": 0.7320467832874744, | |
| "min": -0.9341125530190766, | |
| "max": 1.1131106953003578, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.CumulativeReward.sum": { | |
| "value": 34.4061988145113, | |
| "min": -29.89160169661045, | |
| "max": 62.33419893682003, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicReward.mean": { | |
| "value": 0.7320467832874744, | |
| "min": -0.9341125530190766, | |
| "max": 1.1131106953003578, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicReward.sum": { | |
| "value": 34.4061988145113, | |
| "min": -29.89160169661045, | |
| "max": 62.33419893682003, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndReward.mean": { | |
| "value": 0.17808634871765575, | |
| "min": 0.15359965654418126, | |
| "max": 9.569400081411004, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndReward.sum": { | |
| "value": 8.37005838972982, | |
| "min": 8.064975622051861, | |
| "max": 153.11040130257607, | |
| "count": 33 | |
| }, | |
| "Pyramids.IsTraining.mean": { | |
| "value": 1.0, | |
| "min": 1.0, | |
| "max": 1.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.IsTraining.sum": { | |
| "value": 1.0, | |
| "min": 1.0, | |
| "max": 1.0, | |
| "count": 33 | |
| } | |
| }, | |
| "metadata": { | |
| "timer_format_version": "0.1.0", | |
| "start_time_seconds": "1781877410", | |
| "python_version": "3.10.11 (main, May 16 2023, 00:28:57) [GCC 11.2.0]", | |
| "command_line_arguments": "/usr/local/bin/mlagents-learn ./config/ppo/PyramidsRND.yaml --env=./training-envs-executables/linux/Pyramids/Pyramids --run-id=Pyramids Training --no-graphics --resume", | |
| "mlagents_version": "1.2.0.dev0", | |
| "mlagents_envs_version": "1.2.0.dev0", | |
| "communication_protocol_version": "1.5.0", | |
| "pytorch_version": "2.8.0+cu128", | |
| "numpy_version": "1.23.5", | |
| "end_time_seconds": "1781879735" | |
| }, | |
| "total": 2325.3920998160006, | |
| "count": 1, | |
| "self": 0.47915704200113396, | |
| "children": { | |
| "run_training.setup": { | |
| "total": 0.02286060299957171, | |
| "count": 1, | |
| "self": 0.02286060299957171 | |
| }, | |
| "TrainerController.start_learning": { | |
| "total": 2324.890082171, | |
| "count": 1, | |
| "self": 1.3180784369728826, | |
| "children": { | |
| "TrainerController._reset_env": { | |
| "total": 2.1514990559999205, | |
| "count": 1, | |
| "self": 2.1514990559999205 | |
| }, | |
| "TrainerController.advance": { | |
| "total": 2321.3433243000272, | |
| "count": 62999, | |
| "self": 1.3640972560515365, | |
| "children": { | |
| "env_step": { | |
| "total": 1703.7966754910885, | |
| "count": 62999, | |
| "self": 1552.2669364380813, | |
| "children": { | |
| "SubprocessEnvManager._take_step": { | |
| "total": 150.7262089540136, | |
| "count": 62999, | |
| "self": 4.574016635930093, | |
| "children": { | |
| "TorchPolicy.evaluate": { | |
| "total": 146.1521923180835, | |
| "count": 62165, | |
| "self": 146.1521923180835 | |
| } | |
| } | |
| }, | |
| "workers": { | |
| "total": 0.8035300989936331, | |
| "count": 62999, | |
| "self": 0.0, | |
| "children": { | |
| "worker_root": { | |
| "total": 2319.3881338460696, | |
| "count": 62999, | |
| "is_parallel": true, | |
| "self": 882.3121545960785, | |
| "children": { | |
| "run_training.setup": { | |
| "total": 0.0, | |
| "count": 0, | |
| "is_parallel": true, | |
| "self": 0.0, | |
| "children": { | |
| "steps_from_proto": { | |
| "total": 0.0018806130001394195, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0006665560003966675, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 0.001214056999742752, | |
| "count": 8, | |
| "is_parallel": true, | |
| "self": 0.001214056999742752 | |
| } | |
| } | |
| }, | |
| "UnityEnvironment.step": { | |
| "total": 0.05128879899984895, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0005539949997910298, | |
| "children": { | |
| "UnityEnvironment._generate_step_input": { | |
| "total": 0.0005001140002605098, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0005001140002605098 | |
| }, | |
| "communicator.exchange": { | |
| "total": 0.04863773399983984, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.04863773399983984 | |
| }, | |
| "steps_from_proto": { | |
| "total": 0.0015969559999575722, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.00038586499977100175, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 0.0012110910001865705, | |
| "count": 8, | |
| "is_parallel": true, | |
| "self": 0.0012110910001865705 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "UnityEnvironment.step": { | |
| "total": 1437.0759792499912, | |
| "count": 62998, | |
| "is_parallel": true, | |
| "self": 33.40167111589244, | |
| "children": { | |
| "UnityEnvironment._generate_step_input": { | |
| "total": 23.166133541987165, | |
| "count": 62998, | |
| "is_parallel": true, | |
| "self": 23.166133541987165 | |
| }, | |
| "communicator.exchange": { | |
| "total": 1274.2417032561098, | |
| "count": 62998, | |
| "is_parallel": true, | |
| "self": 1274.2417032561098 | |
| }, | |
| "steps_from_proto": { | |
| "total": 106.26647133600181, | |
| "count": 62998, | |
| "is_parallel": true, | |
| "self": 22.144776773895046, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 84.12169456210677, | |
| "count": 503984, | |
| "is_parallel": true, | |
| "self": 84.12169456210677 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "trainer_advance": { | |
| "total": 616.1825515528872, | |
| "count": 62999, | |
| "self": 2.5511752717175114, | |
| "children": { | |
| "process_trajectory": { | |
| "total": 109.11603575416211, | |
| "count": 62999, | |
| "self": 108.92333363416265, | |
| "children": { | |
| "RLTrainer._checkpoint": { | |
| "total": 0.19270211999946696, | |
| "count": 2, | |
| "self": 0.19270211999946696 | |
| } | |
| } | |
| }, | |
| "_update_policy": { | |
| "total": 504.51534052700754, | |
| "count": 452, | |
| "self": 266.3646412810058, | |
| "children": { | |
| "TorchPPOOptimizer.update": { | |
| "total": 238.15069924600175, | |
| "count": 22662, | |
| "self": 238.15069924600175 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "trainer_threads": { | |
| "total": 1.0900002962443978e-06, | |
| "count": 1, | |
| "self": 1.0900002962443978e-06 | |
| }, | |
| "TrainerController._save_models": { | |
| "total": 0.07717928799957008, | |
| "count": 1, | |
| "self": 0.0014161479994072579, | |
| "children": { | |
| "RLTrainer._checkpoint": { | |
| "total": 0.07576314000016282, | |
| "count": 1, | |
| "self": 0.07576314000016282 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } |