Reinforcement Learning
ml-agents
TensorBoard
ONNX
Pyramids
deep-reinforcement-learning
ML-Agents-Pyramids
Instructions to use Yashwanthkumar18/ppo-Pyramids with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- ml-agents
How to use Yashwanthkumar18/ppo-Pyramids with ml-agents:
mlagents-load-from-hf --repo-id="Yashwanthkumar18/ppo-Pyramids" --local-dir="./download: string[]s"
- Notebooks
- Google Colab
- Kaggle
| { | |
| "name": "root", | |
| "gauges": { | |
| "Pyramids.Policy.Entropy.mean": { | |
| "value": 0.48783212900161743, | |
| "min": 0.4809330403804779, | |
| "max": 1.4088127613067627, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Entropy.sum": { | |
| "value": 14713.0166015625, | |
| "min": 14528.025390625, | |
| "max": 42737.7421875, | |
| "count": 33 | |
| }, | |
| "Pyramids.Step.mean": { | |
| "value": 989952.0, | |
| "min": 29954.0, | |
| "max": 989952.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Step.sum": { | |
| "value": 989952.0, | |
| "min": 29954.0, | |
| "max": 989952.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicValueEstimate.mean": { | |
| "value": 0.3786240518093109, | |
| "min": -0.09374184161424637, | |
| "max": 0.3786240518093109, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicValueEstimate.sum": { | |
| "value": 99.95674896240234, | |
| "min": -22.4980411529541, | |
| "max": 99.95674896240234, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndValueEstimate.mean": { | |
| "value": 0.014619356952607632, | |
| "min": -0.0008249550010077655, | |
| "max": 0.32800212502479553, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndValueEstimate.sum": { | |
| "value": 3.8595101833343506, | |
| "min": -0.21531325578689575, | |
| "max": 78.72051239013672, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.PolicyLoss.mean": { | |
| "value": 0.06822947945260067, | |
| "min": 0.06489771259740965, | |
| "max": 0.07349181617724289, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.PolicyLoss.sum": { | |
| "value": 0.9552127123364094, | |
| "min": 0.4817497198894451, | |
| "max": 1.045439059315957, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.ValueLoss.mean": { | |
| "value": 0.010695153003325686, | |
| "min": 0.0006909796061438143, | |
| "max": 0.011399863236551261, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.ValueLoss.sum": { | |
| "value": 0.1497321420465596, | |
| "min": 0.0096737144860134, | |
| "max": 0.15986601463459119, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.LearningRate.mean": { | |
| "value": 7.518076065435716e-06, | |
| "min": 7.518076065435716e-06, | |
| "max": 0.00029515063018788575, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.LearningRate.sum": { | |
| "value": 0.00010525306491610003, | |
| "min": 0.00010525306491610003, | |
| "max": 0.0035079428306857992, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Epsilon.mean": { | |
| "value": 0.10250599285714287, | |
| "min": 0.10250599285714287, | |
| "max": 0.19838354285714285, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Epsilon.sum": { | |
| "value": 1.4350839000000002, | |
| "min": 1.3886848, | |
| "max": 2.5693142000000004, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Beta.mean": { | |
| "value": 0.0002603486864285715, | |
| "min": 0.0002603486864285715, | |
| "max": 0.00983851593142857, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Beta.sum": { | |
| "value": 0.003644881610000001, | |
| "min": 0.003644881610000001, | |
| "max": 0.11695448858000002, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.RNDLoss.mean": { | |
| "value": 0.011734756641089916, | |
| "min": 0.011646255850791931, | |
| "max": 0.4189160466194153, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.RNDLoss.sum": { | |
| "value": 0.16428659856319427, | |
| "min": 0.16304758191108704, | |
| "max": 2.9324123859405518, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.EpisodeLength.mean": { | |
| "value": 487.4754098360656, | |
| "min": 487.4754098360656, | |
| "max": 999.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.EpisodeLength.sum": { | |
| "value": 29736.0, | |
| "min": 16753.0, | |
| "max": 32837.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.CumulativeReward.mean": { | |
| "value": 1.252190297529582, | |
| "min": -0.9999806972280625, | |
| "max": 1.252190297529582, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.CumulativeReward.sum": { | |
| "value": 77.63579844683409, | |
| "min": -30.99940161406994, | |
| "max": 77.63579844683409, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicReward.mean": { | |
| "value": 1.252190297529582, | |
| "min": -0.9999806972280625, | |
| "max": 1.252190297529582, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicReward.sum": { | |
| "value": 77.63579844683409, | |
| "min": -30.99940161406994, | |
| "max": 77.63579844683409, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndReward.mean": { | |
| "value": 0.05915487584500243, | |
| "min": 0.05915487584500243, | |
| "max": 7.5946758096380265, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndReward.sum": { | |
| "value": 3.6676023023901507, | |
| "min": 3.3404390568030067, | |
| "max": 129.10948876384646, | |
| "count": 33 | |
| }, | |
| "Pyramids.IsTraining.mean": { | |
| "value": 1.0, | |
| "min": 1.0, | |
| "max": 1.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.IsTraining.sum": { | |
| "value": 1.0, | |
| "min": 1.0, | |
| "max": 1.0, | |
| "count": 33 | |
| } | |
| }, | |
| "metadata": { | |
| "timer_format_version": "0.1.0", | |
| "start_time_seconds": "1788702661", | |
| "python_version": "3.10.12 (main, Jul 5 2023, 18:54:27) [GCC 11.2.0]", | |
| "command_line_arguments": "/usr/local/envs/mlagents_env/bin/mlagents-learn ./ml-agents/config/ppo/PyramidsRND.yaml --env=./training-envs-executables/linux/Pyramids/Pyramids --run-id=Pyramids_Training --no-graphics", | |
| "mlagents_version": "1.2.0.dev0", | |
| "mlagents_envs_version": "1.2.0.dev0", | |
| "communication_protocol_version": "1.5.0", | |
| "pytorch_version": "2.8.0+cu128", | |
| "numpy_version": "1.23.5", | |
| "end_time_seconds": "1788704966" | |
| }, | |
| "total": 2305.0791632019996, | |
| "count": 1, | |
| "self": 0.4791425659991546, | |
| "children": { | |
| "run_training.setup": { | |
| "total": 0.0288110719998258, | |
| "count": 1, | |
| "self": 0.0288110719998258 | |
| }, | |
| "TrainerController.start_learning": { | |
| "total": 2304.5712095640006, | |
| "count": 1, | |
| "self": 1.3243847240287323, | |
| "children": { | |
| "TrainerController._reset_env": { | |
| "total": 2.157245496000087, | |
| "count": 1, | |
| "self": 2.157245496000087 | |
| }, | |
| "TrainerController.advance": { | |
| "total": 2301.0036923229713, | |
| "count": 63454, | |
| "self": 1.3941105970220633, | |
| "children": { | |
| "env_step": { | |
| "total": 1672.2148914779605, | |
| "count": 63454, | |
| "self": 1516.6525901740001, | |
| "children": { | |
| "SubprocessEnvManager._take_step": { | |
| "total": 154.74295719802512, | |
| "count": 63454, | |
| "self": 4.642935758044132, | |
| "children": { | |
| "TorchPolicy.evaluate": { | |
| "total": 150.100021439981, | |
| "count": 62564, | |
| "self": 150.100021439981 | |
| } | |
| } | |
| }, | |
| "workers": { | |
| "total": 0.8193441059352153, | |
| "count": 63454, | |
| "self": 0.0, | |
| "children": { | |
| "worker_root": { | |
| "total": 2298.5768821619226, | |
| "count": 63454, | |
| "is_parallel": true, | |
| "self": 899.2950478037978, | |
| "children": { | |
| "run_training.setup": { | |
| "total": 0.0, | |
| "count": 0, | |
| "is_parallel": true, | |
| "self": 0.0, | |
| "children": { | |
| "steps_from_proto": { | |
| "total": 0.0018205340002168668, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0005613369999082352, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 0.0012591970003086317, | |
| "count": 8, | |
| "is_parallel": true, | |
| "self": 0.0012591970003086317 | |
| } | |
| } | |
| }, | |
| "UnityEnvironment.step": { | |
| "total": 0.055983178000133194, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0006176660003802681, | |
| "children": { | |
| "UnityEnvironment._generate_step_input": { | |
| "total": 0.0005141530000400962, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0005141530000400962 | |
| }, | |
| "communicator.exchange": { | |
| "total": 0.05299211200008358, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.05299211200008358 | |
| }, | |
| "steps_from_proto": { | |
| "total": 0.0018592469996292493, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0003831099988929054, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 0.001476137000736344, | |
| "count": 8, | |
| "is_parallel": true, | |
| "self": 0.001476137000736344 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "UnityEnvironment.step": { | |
| "total": 1399.2818343581248, | |
| "count": 63453, | |
| "is_parallel": true, | |
| "self": 34.19844318621108, | |
| "children": { | |
| "UnityEnvironment._generate_step_input": { | |
| "total": 24.08182291593903, | |
| "count": 63453, | |
| "is_parallel": true, | |
| "self": 24.08182291593903 | |
| }, | |
| "communicator.exchange": { | |
| "total": 1230.437334018995, | |
| "count": 63453, | |
| "is_parallel": true, | |
| "self": 1230.437334018995 | |
| }, | |
| "steps_from_proto": { | |
| "total": 110.5642342369797, | |
| "count": 63453, | |
| "is_parallel": true, | |
| "self": 22.99906075076842, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 87.56517348621128, | |
| "count": 507624, | |
| "is_parallel": true, | |
| "self": 87.56517348621128 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "trainer_advance": { | |
| "total": 627.3946902479888, | |
| "count": 63454, | |
| "self": 2.6929361549173336, | |
| "children": { | |
| "process_trajectory": { | |
| "total": 109.9341929870684, | |
| "count": 63454, | |
| "self": 109.74110866706769, | |
| "children": { | |
| "RLTrainer._checkpoint": { | |
| "total": 0.19308432000070752, | |
| "count": 2, | |
| "self": 0.19308432000070752 | |
| } | |
| } | |
| }, | |
| "_update_policy": { | |
| "total": 514.7675611060031, | |
| "count": 450, | |
| "self": 275.2608939550014, | |
| "children": { | |
| "TorchPPOOptimizer.update": { | |
| "total": 239.5066671510017, | |
| "count": 22740, | |
| "self": 239.5066671510017 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "trainer_threads": { | |
| "total": 8.990000424091704e-07, | |
| "count": 1, | |
| "self": 8.990000424091704e-07 | |
| }, | |
| "TrainerController._save_models": { | |
| "total": 0.0858861220003746, | |
| "count": 1, | |
| "self": 0.0010058050002044183, | |
| "children": { | |
| "RLTrainer._checkpoint": { | |
| "total": 0.08488031700017018, | |
| "count": 1, | |
| "self": 0.08488031700017018 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } |