Reinforcement Learning
ml-agents
TensorBoard
ONNX
Pyramids
deep-reinforcement-learning
ML-Agents-Pyramids
Instructions to use ziodraw/ppo-PyramidsTraining with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- ml-agents
How to use ziodraw/ppo-PyramidsTraining with ml-agents:
mlagents-load-from-hf --repo-id="ziodraw/ppo-PyramidsTraining" --local-dir="./download: string[]s"
- Notebooks
- Google Colab
- Kaggle
| { | |
| "name": "root", | |
| "gauges": { | |
| "Pyramids.Policy.Entropy.mean": { | |
| "value": 0.4509788155555725, | |
| "min": 0.4509788155555725, | |
| "max": 1.4003734588623047, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Entropy.sum": { | |
| "value": 13536.580078125, | |
| "min": 13444.806640625, | |
| "max": 42481.73046875, | |
| "count": 33 | |
| }, | |
| "Pyramids.Step.mean": { | |
| "value": 989999.0, | |
| "min": 29952.0, | |
| "max": 989999.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Step.sum": { | |
| "value": 989999.0, | |
| "min": 29952.0, | |
| "max": 989999.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicValueEstimate.mean": { | |
| "value": 0.3352579176425934, | |
| "min": -0.08843117207288742, | |
| "max": 0.4124199151992798, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicValueEstimate.sum": { | |
| "value": 87.5023193359375, | |
| "min": -21.223482131958008, | |
| "max": 112.59063720703125, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndValueEstimate.mean": { | |
| "value": -0.102295882999897, | |
| "min": -0.102295882999897, | |
| "max": 0.4489445090293884, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndValueEstimate.sum": { | |
| "value": -26.69922637939453, | |
| "min": -26.69922637939453, | |
| "max": 107.7466812133789, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.PolicyLoss.mean": { | |
| "value": 0.0693691586135627, | |
| "min": 0.06440655503355826, | |
| "max": 0.07240279649700877, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.PolicyLoss.sum": { | |
| "value": 0.9711682205898778, | |
| "min": 0.5000358172364522, | |
| "max": 1.0682327381287264, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.ValueLoss.mean": { | |
| "value": 0.01904471817186338, | |
| "min": 0.0007915219617923386, | |
| "max": 0.01904471817186338, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.ValueLoss.sum": { | |
| "value": 0.26662605440608733, | |
| "min": 0.006332175694338709, | |
| "max": 0.26662605440608733, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.LearningRate.mean": { | |
| "value": 7.428333238207143e-06, | |
| "min": 7.428333238207143e-06, | |
| "max": 0.00029515063018788575, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.LearningRate.sum": { | |
| "value": 0.0001039966653349, | |
| "min": 0.0001039966653349, | |
| "max": 0.0033752629749124, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Epsilon.mean": { | |
| "value": 0.10247607857142858, | |
| "min": 0.10247607857142858, | |
| "max": 0.19838354285714285, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Epsilon.sum": { | |
| "value": 1.4346651000000001, | |
| "min": 1.3691136000000002, | |
| "max": 2.4846914000000004, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Beta.mean": { | |
| "value": 0.0002573602492857143, | |
| "min": 0.0002573602492857143, | |
| "max": 0.00983851593142857, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Beta.sum": { | |
| "value": 0.0036030434900000004, | |
| "min": 0.0036030434900000004, | |
| "max": 0.11252625123999999, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.RNDLoss.mean": { | |
| "value": 0.01707293465733528, | |
| "min": 0.016632163897156715, | |
| "max": 0.7517451047897339, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.RNDLoss.sum": { | |
| "value": 0.23902109265327454, | |
| "min": 0.23285029828548431, | |
| "max": 5.262215614318848, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.EpisodeLength.mean": { | |
| "value": 463.92424242424244, | |
| "min": 425.7042253521127, | |
| "max": 999.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.EpisodeLength.sum": { | |
| "value": 30619.0, | |
| "min": 15984.0, | |
| "max": 33875.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.CumulativeReward.mean": { | |
| "value": 1.2731817962996888, | |
| "min": -1.0000000521540642, | |
| "max": 1.4615887097069915, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.CumulativeReward.sum": { | |
| "value": 84.02999855577946, | |
| "min": -32.000001668930054, | |
| "max": 103.7727983891964, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicReward.mean": { | |
| "value": 1.2731817962996888, | |
| "min": -1.0000000521540642, | |
| "max": 1.4615887097069915, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicReward.sum": { | |
| "value": 84.02999855577946, | |
| "min": -32.000001668930054, | |
| "max": 103.7727983891964, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndReward.mean": { | |
| "value": 0.08077332623758014, | |
| "min": 0.07736101457160491, | |
| "max": 14.34756095521152, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndReward.sum": { | |
| "value": 5.33103953168029, | |
| "min": 5.138807103605359, | |
| "max": 229.56097528338432, | |
| "count": 33 | |
| }, | |
| "Pyramids.IsTraining.mean": { | |
| "value": 1.0, | |
| "min": 1.0, | |
| "max": 1.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.IsTraining.sum": { | |
| "value": 1.0, | |
| "min": 1.0, | |
| "max": 1.0, | |
| "count": 33 | |
| } | |
| }, | |
| "metadata": { | |
| "timer_format_version": "0.1.0", | |
| "start_time_seconds": "1781073957", | |
| "python_version": "3.10.12 | packaged by conda-forge | (main, Jun 23 2023, 22:40:32) [GCC 12.3.0]", | |
| "command_line_arguments": "/usr/local/bin/mlagents-learn ./config/ppo/PyramidsRND.yaml --env=./training-envs-executables/linux/Pyramids/Pyramids --run-id=Pyramids Training --no-graphics", | |
| "mlagents_version": "1.2.0.dev0", | |
| "mlagents_envs_version": "1.2.0.dev0", | |
| "communication_protocol_version": "1.5.0", | |
| "pytorch_version": "2.8.0+cu128", | |
| "numpy_version": "1.23.5", | |
| "end_time_seconds": "1781076502" | |
| }, | |
| "total": 2545.292901467, | |
| "count": 1, | |
| "self": 0.8441568699995514, | |
| "children": { | |
| "run_training.setup": { | |
| "total": 0.02469042500024443, | |
| "count": 1, | |
| "self": 0.02469042500024443 | |
| }, | |
| "TrainerController.start_learning": { | |
| "total": 2544.424054172, | |
| "count": 1, | |
| "self": 1.6330492698812122, | |
| "children": { | |
| "TrainerController._reset_env": { | |
| "total": 2.189360659000158, | |
| "count": 1, | |
| "self": 2.189360659000158 | |
| }, | |
| "TrainerController.advance": { | |
| "total": 2540.4677484661192, | |
| "count": 63574, | |
| "self": 1.619265965059185, | |
| "children": { | |
| "env_step": { | |
| "total": 1842.7876201809731, | |
| "count": 63574, | |
| "self": 1671.0502203580218, | |
| "children": { | |
| "SubprocessEnvManager._take_step": { | |
| "total": 170.7936539319985, | |
| "count": 63574, | |
| "self": 5.095862474003297, | |
| "children": { | |
| "TorchPolicy.evaluate": { | |
| "total": 165.6977914579952, | |
| "count": 62576, | |
| "self": 165.6977914579952 | |
| } | |
| } | |
| }, | |
| "workers": { | |
| "total": 0.9437458909528686, | |
| "count": 63574, | |
| "self": 0.0, | |
| "children": { | |
| "worker_root": { | |
| "total": 2538.45189064405, | |
| "count": 63574, | |
| "is_parallel": true, | |
| "self": 996.3487142880485, | |
| "children": { | |
| "run_training.setup": { | |
| "total": 0.0, | |
| "count": 0, | |
| "is_parallel": true, | |
| "self": 0.0, | |
| "children": { | |
| "steps_from_proto": { | |
| "total": 0.001954257999841502, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0007278459997905884, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 0.0012264120000509138, | |
| "count": 8, | |
| "is_parallel": true, | |
| "self": 0.0012264120000509138 | |
| } | |
| } | |
| }, | |
| "UnityEnvironment.step": { | |
| "total": 0.05152630299971861, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0005720809995182208, | |
| "children": { | |
| "UnityEnvironment._generate_step_input": { | |
| "total": 0.00042921000022033695, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.00042921000022033695 | |
| }, | |
| "communicator.exchange": { | |
| "total": 0.0487769629999093, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0487769629999093 | |
| }, | |
| "steps_from_proto": { | |
| "total": 0.0017480490000707505, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.00038666300042677904, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 0.0013613859996439714, | |
| "count": 8, | |
| "is_parallel": true, | |
| "self": 0.0013613859996439714 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "UnityEnvironment.step": { | |
| "total": 1542.1031763560013, | |
| "count": 63573, | |
| "is_parallel": true, | |
| "self": 34.843647935070294, | |
| "children": { | |
| "UnityEnvironment._generate_step_input": { | |
| "total": 24.797576023961938, | |
| "count": 63573, | |
| "is_parallel": true, | |
| "self": 24.797576023961938 | |
| }, | |
| "communicator.exchange": { | |
| "total": 1364.9630108918768, | |
| "count": 63573, | |
| "is_parallel": true, | |
| "self": 1364.9630108918768 | |
| }, | |
| "steps_from_proto": { | |
| "total": 117.49894150509226, | |
| "count": 63573, | |
| "is_parallel": true, | |
| "self": 24.607476493179547, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 92.89146501191271, | |
| "count": 508584, | |
| "is_parallel": true, | |
| "self": 92.89146501191271 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "trainer_advance": { | |
| "total": 696.0608623200869, | |
| "count": 63574, | |
| "self": 3.150574159111329, | |
| "children": { | |
| "process_trajectory": { | |
| "total": 134.56850047598118, | |
| "count": 63574, | |
| "self": 134.32755888198153, | |
| "children": { | |
| "RLTrainer._checkpoint": { | |
| "total": 0.24094159399965065, | |
| "count": 2, | |
| "self": 0.24094159399965065 | |
| } | |
| } | |
| }, | |
| "_update_policy": { | |
| "total": 558.3417876849944, | |
| "count": 444, | |
| "self": 307.10422799091793, | |
| "children": { | |
| "TorchPPOOptimizer.update": { | |
| "total": 251.23755969407648, | |
| "count": 22821, | |
| "self": 251.23755969407648 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "trainer_threads": { | |
| "total": 1.7060001482605003e-06, | |
| "count": 1, | |
| "self": 1.7060001482605003e-06 | |
| }, | |
| "TrainerController._save_models": { | |
| "total": 0.13389407099930395, | |
| "count": 1, | |
| "self": 0.0014610659991376451, | |
| "children": { | |
| "RLTrainer._checkpoint": { | |
| "total": 0.1324330050001663, | |
| "count": 1, | |
| "self": 0.1324330050001663 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } |