Reinforcement Learning
ml-agents
TensorBoard
ONNX
Pyramids
deep-reinforcement-learning
ML-Agents-Pyramids
Instructions to use LATlag/ppo-Pyramids with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- ml-agents
How to use LATlag/ppo-Pyramids with ml-agents:
mlagents-load-from-hf --repo-id="LATlag/ppo-Pyramids" --local-dir="./download: string[]s"
- Notebooks
- Google Colab
- Kaggle
| { | |
| "name": "root", | |
| "gauges": { | |
| "Pyramids.Policy.Entropy.mean": { | |
| "value": 0.3502959609031677, | |
| "min": 0.3502959609031677, | |
| "max": 1.4698705673217773, | |
| "count": 28 | |
| }, | |
| "Pyramids.Policy.Entropy.sum": { | |
| "value": 10525.693359375, | |
| "min": 10525.693359375, | |
| "max": 44589.9921875, | |
| "count": 28 | |
| }, | |
| "Pyramids.Step.mean": { | |
| "value": 839989.0, | |
| "min": 29923.0, | |
| "max": 839989.0, | |
| "count": 28 | |
| }, | |
| "Pyramids.Step.sum": { | |
| "value": 839989.0, | |
| "min": 29923.0, | |
| "max": 839989.0, | |
| "count": 28 | |
| }, | |
| "Pyramids.Policy.ExtrinsicValueEstimate.mean": { | |
| "value": 0.629786491394043, | |
| "min": -0.09799198806285858, | |
| "max": 0.629786491394043, | |
| "count": 28 | |
| }, | |
| "Pyramids.Policy.ExtrinsicValueEstimate.sum": { | |
| "value": 176.34022521972656, | |
| "min": -23.616069793701172, | |
| "max": 176.34022521972656, | |
| "count": 28 | |
| }, | |
| "Pyramids.Policy.RndValueEstimate.mean": { | |
| "value": 0.004649701062589884, | |
| "min": -0.013009719550609589, | |
| "max": 0.26096516847610474, | |
| "count": 28 | |
| }, | |
| "Pyramids.Policy.RndValueEstimate.sum": { | |
| "value": 1.301916241645813, | |
| "min": -3.473595142364502, | |
| "max": 62.89260482788086, | |
| "count": 28 | |
| }, | |
| "Pyramids.Losses.PolicyLoss.mean": { | |
| "value": 0.06638622395569525, | |
| "min": 0.06423629471100867, | |
| "max": 0.0742484789061596, | |
| "count": 28 | |
| }, | |
| "Pyramids.Losses.PolicyLoss.sum": { | |
| "value": 0.9294071353797335, | |
| "min": 0.5012575005570523, | |
| "max": 1.0661113334045855, | |
| "count": 28 | |
| }, | |
| "Pyramids.Losses.ValueLoss.mean": { | |
| "value": 0.01573695085660022, | |
| "min": 0.0005568967800390506, | |
| "max": 0.015971414667488328, | |
| "count": 28 | |
| }, | |
| "Pyramids.Losses.ValueLoss.sum": { | |
| "value": 0.2203173119924031, | |
| "min": 0.007239658140507658, | |
| "max": 0.2235998053448366, | |
| "count": 28 | |
| }, | |
| "Pyramids.Policy.LearningRate.mean": { | |
| "value": 0.0002175689131913238, | |
| "min": 0.0002175689131913238, | |
| "max": 0.0002984113862438238, | |
| "count": 28 | |
| }, | |
| "Pyramids.Policy.LearningRate.sum": { | |
| "value": 0.003045964784678533, | |
| "min": 0.0020888797037067666, | |
| "max": 0.004027256057581366, | |
| "count": 28 | |
| }, | |
| "Pyramids.Policy.Epsilon.mean": { | |
| "value": 0.1725229619047619, | |
| "min": 0.1725229619047619, | |
| "max": 0.19947046190476195, | |
| "count": 28 | |
| }, | |
| "Pyramids.Policy.Epsilon.sum": { | |
| "value": 2.4153214666666667, | |
| "min": 1.3962932333333335, | |
| "max": 2.8424186333333337, | |
| "count": 28 | |
| }, | |
| "Pyramids.Policy.Beta.mean": { | |
| "value": 0.007255043894285713, | |
| "min": 0.007255043894285713, | |
| "max": 0.009947099144285713, | |
| "count": 28 | |
| }, | |
| "Pyramids.Policy.Beta.sum": { | |
| "value": 0.10157061451999999, | |
| "min": 0.06962969400999999, | |
| "max": 0.13425762147, | |
| "count": 28 | |
| }, | |
| "Pyramids.Losses.RNDLoss.mean": { | |
| "value": 0.013835138641297817, | |
| "min": 0.01322467066347599, | |
| "max": 0.3313908576965332, | |
| "count": 28 | |
| }, | |
| "Pyramids.Losses.RNDLoss.sum": { | |
| "value": 0.1936919391155243, | |
| "min": 0.18514539301395416, | |
| "max": 2.3197360038757324, | |
| "count": 28 | |
| }, | |
| "Pyramids.Environment.EpisodeLength.mean": { | |
| "value": 313.3369565217391, | |
| "min": 313.3369565217391, | |
| "max": 992.875, | |
| "count": 28 | |
| }, | |
| "Pyramids.Environment.EpisodeLength.sum": { | |
| "value": 28827.0, | |
| "min": 16594.0, | |
| "max": 32929.0, | |
| "count": 28 | |
| }, | |
| "Pyramids.Environment.CumulativeReward.mean": { | |
| "value": 1.5561782439117846, | |
| "min": -0.9310187993105501, | |
| "max": 1.6087899799458683, | |
| "count": 28 | |
| }, | |
| "Pyramids.Environment.CumulativeReward.sum": { | |
| "value": 143.1683984398842, | |
| "min": -30.246001705527306, | |
| "max": 144.8747985586524, | |
| "count": 28 | |
| }, | |
| "Pyramids.Policy.ExtrinsicReward.mean": { | |
| "value": 1.5561782439117846, | |
| "min": -0.9310187993105501, | |
| "max": 1.6087899799458683, | |
| "count": 28 | |
| }, | |
| "Pyramids.Policy.ExtrinsicReward.sum": { | |
| "value": 143.1683984398842, | |
| "min": -30.246001705527306, | |
| "max": 144.8747985586524, | |
| "count": 28 | |
| }, | |
| "Pyramids.Policy.RndReward.mean": { | |
| "value": 0.04438819246412174, | |
| "min": 0.04438819246412174, | |
| "max": 5.594401250867283, | |
| "count": 28 | |
| }, | |
| "Pyramids.Policy.RndReward.sum": { | |
| "value": 4.0837137066992, | |
| "min": 4.037327836878831, | |
| "max": 95.1048212647438, | |
| "count": 28 | |
| }, | |
| "Pyramids.IsTraining.mean": { | |
| "value": 1.0, | |
| "min": 1.0, | |
| "max": 1.0, | |
| "count": 28 | |
| }, | |
| "Pyramids.IsTraining.sum": { | |
| "value": 1.0, | |
| "min": 1.0, | |
| "max": 1.0, | |
| "count": 28 | |
| } | |
| }, | |
| "metadata": { | |
| "timer_format_version": "0.1.0", | |
| "start_time_seconds": "1786768430", | |
| "python_version": "3.10.12 (main, Jul 5 2023, 18:54:27) [GCC 11.2.0]", | |
| "command_line_arguments": "/home/lagnesh/Projects/DeepRL/ml-agents/ml-agents/mlagents/trainers/learn.py /home/lagnesh/Projects/DeepRL/ml-agents/config/ppo/PyramidsRND.yaml --env=./Pyramids_env/Pyramids/Pyramids --run-id=PyramidsTraining1 --no-graphics --force", | |
| "mlagents_version": "1.2.0.dev0", | |
| "mlagents_envs_version": "1.2.0.dev0", | |
| "communication_protocol_version": "1.5.0", | |
| "pytorch_version": "2.8.0+cu128", | |
| "numpy_version": "1.23.5", | |
| "end_time_seconds": "1786771356" | |
| }, | |
| "total": 2795.8748361240005, | |
| "count": 1, | |
| "self": 3.5503616580008384, | |
| "children": { | |
| "run_training.setup": { | |
| "total": 0.017944634999821574, | |
| "count": 1, | |
| "self": 0.017944634999821574 | |
| }, | |
| "TrainerController.start_learning": { | |
| "total": 2792.306529831, | |
| "count": 1, | |
| "self": 1.788223593083785, | |
| "children": { | |
| "TrainerController._reset_env": { | |
| "total": 3.876147533999756, | |
| "count": 1, | |
| "self": 3.876147533999756 | |
| }, | |
| "TrainerController.advance": { | |
| "total": 2785.636464730915, | |
| "count": 54262, | |
| "self": 1.963086632825707, | |
| "children": { | |
| "env_step": { | |
| "total": 1893.3336433690815, | |
| "count": 54262, | |
| "self": 1582.394395377144, | |
| "children": { | |
| "SubprocessEnvManager._take_step": { | |
| "total": 309.75291562593065, | |
| "count": 54262, | |
| "self": 7.058743866009536, | |
| "children": { | |
| "TorchPolicy.evaluate": { | |
| "total": 302.6941717599211, | |
| "count": 53113, | |
| "self": 302.6941717599211 | |
| } | |
| } | |
| }, | |
| "workers": { | |
| "total": 1.186332366006809, | |
| "count": 54261, | |
| "self": 0.0, | |
| "children": { | |
| "worker_root": { | |
| "total": 2787.210104066988, | |
| "count": 54261, | |
| "is_parallel": true, | |
| "self": 1328.0417568640319, | |
| "children": { | |
| "run_training.setup": { | |
| "total": 0.0, | |
| "count": 0, | |
| "is_parallel": true, | |
| "self": 0.0, | |
| "children": { | |
| "steps_from_proto": { | |
| "total": 0.001808707000236609, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0009546139999656589, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 0.0008540930002709501, | |
| "count": 8, | |
| "is_parallel": true, | |
| "self": 0.0008540930002709501 | |
| } | |
| } | |
| }, | |
| "UnityEnvironment.step": { | |
| "total": 0.03315804500016384, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.00020833999997194041, | |
| "children": { | |
| "UnityEnvironment._generate_step_input": { | |
| "total": 0.00021028000037404126, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.00021028000037404126 | |
| }, | |
| "communicator.exchange": { | |
| "total": 0.03215912499990736, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.03215912499990736 | |
| }, | |
| "steps_from_proto": { | |
| "total": 0.0005802999999104941, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0001609020005162165, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 0.0004193979993942776, | |
| "count": 8, | |
| "is_parallel": true, | |
| "self": 0.0004193979993942776 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "UnityEnvironment.step": { | |
| "total": 1459.1683472029563, | |
| "count": 54260, | |
| "is_parallel": true, | |
| "self": 20.472134827994523, | |
| "children": { | |
| "UnityEnvironment._generate_step_input": { | |
| "total": 16.000049496893098, | |
| "count": 54260, | |
| "is_parallel": true, | |
| "self": 16.000049496893098 | |
| }, | |
| "communicator.exchange": { | |
| "total": 1367.3548614210931, | |
| "count": 54260, | |
| "is_parallel": true, | |
| "self": 1367.3548614210931 | |
| }, | |
| "steps_from_proto": { | |
| "total": 55.34130145697554, | |
| "count": 54260, | |
| "is_parallel": true, | |
| "self": 13.656972305909221, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 41.68432915106632, | |
| "count": 434080, | |
| "is_parallel": true, | |
| "self": 41.68432915106632 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "trainer_advance": { | |
| "total": 890.3397347290079, | |
| "count": 54261, | |
| "self": 3.6681823540043297, | |
| "children": { | |
| "process_trajectory": { | |
| "total": 145.49038223899788, | |
| "count": 54261, | |
| "self": 144.04397876999792, | |
| "children": { | |
| "RLTrainer._checkpoint": { | |
| "total": 1.4464034689999608, | |
| "count": 1, | |
| "self": 1.4464034689999608 | |
| } | |
| } | |
| }, | |
| "_update_policy": { | |
| "total": 741.1811701360057, | |
| "count": 385, | |
| "self": 312.1338401250596, | |
| "children": { | |
| "TorchPPOOptimizer.update": { | |
| "total": 429.0473300109461, | |
| "count": 19320, | |
| "self": 429.0473300109461 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "trainer_threads": { | |
| "total": 1.7340007616439834e-06, | |
| "count": 1, | |
| "self": 1.7340007616439834e-06 | |
| }, | |
| "TrainerController._save_models": { | |
| "total": 1.005692239000382, | |
| "count": 1, | |
| "self": 0.0028390030001901323, | |
| "children": { | |
| "RLTrainer._checkpoint": { | |
| "total": 1.0028532360001918, | |
| "count": 1, | |
| "self": 1.0028532360001918 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } |