Reinforcement Learning
ml-agents
TensorBoard
ONNX
Pyramids
deep-reinforcement-learning
ML-Agents-Pyramids
Instructions to use Atharva1232/ppo-Pyramids with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- ml-agents
How to use Atharva1232/ppo-Pyramids with ml-agents:
mlagents-load-from-hf --repo-id="Atharva1232/ppo-Pyramids" --local-dir="./download: string[]s"
- Notebooks
- Google Colab
- Kaggle
| { | |
| "name": "root", | |
| "gauges": { | |
| "Pyramids.Policy.Entropy.mean": { | |
| "value": 0.551590085029602, | |
| "min": 0.551590085029602, | |
| "max": 1.577091097831726, | |
| "count": 50 | |
| }, | |
| "Pyramids.Policy.Entropy.sum": { | |
| "value": 10987.6748046875, | |
| "min": 10987.6748046875, | |
| "max": 31693.22265625, | |
| "count": 50 | |
| }, | |
| "Pyramids.Step.mean": { | |
| "value": 999996.0, | |
| "min": 19968.0, | |
| "max": 999996.0, | |
| "count": 50 | |
| }, | |
| "Pyramids.Step.sum": { | |
| "value": 999996.0, | |
| "min": 19968.0, | |
| "max": 999996.0, | |
| "count": 50 | |
| }, | |
| "Pyramids.Policy.ExtrinsicValueEstimate.mean": { | |
| "value": 0.5138071179389954, | |
| "min": -0.1796068400144577, | |
| "max": 0.5138071179389954, | |
| "count": 50 | |
| }, | |
| "Pyramids.Policy.ExtrinsicValueEstimate.sum": { | |
| "value": 95.56812286376953, | |
| "min": -28.55748748779297, | |
| "max": 95.56812286376953, | |
| "count": 50 | |
| }, | |
| "Pyramids.Policy.RndValueEstimate.mean": { | |
| "value": 0.06619288772344589, | |
| "min": -0.0223530363291502, | |
| "max": 0.21983985602855682, | |
| "count": 50 | |
| }, | |
| "Pyramids.Policy.RndValueEstimate.sum": { | |
| "value": 12.311877250671387, | |
| "min": -3.9788405895233154, | |
| "max": 34.95453643798828, | |
| "count": 50 | |
| }, | |
| "Pyramids.Losses.PolicyLoss.mean": { | |
| "value": 0.06523274813338906, | |
| "min": 0.06457026638263078, | |
| "max": 0.07381447452698248, | |
| "count": 50 | |
| }, | |
| "Pyramids.Losses.PolicyLoss.sum": { | |
| "value": 0.5870947332005015, | |
| "min": 0.29177958178822344, | |
| "max": 0.7300669412795161, | |
| "count": 50 | |
| }, | |
| "Pyramids.Losses.ValueLoss.mean": { | |
| "value": 0.016980483297461265, | |
| "min": 0.0001669898494954476, | |
| "max": 0.017262498813107853, | |
| "count": 50 | |
| }, | |
| "Pyramids.Losses.ValueLoss.sum": { | |
| "value": 0.1528243496771514, | |
| "min": 0.0013359187959635808, | |
| "max": 0.1607116995876947, | |
| "count": 50 | |
| }, | |
| "Pyramids.Policy.LearningRate.mean": { | |
| "value": 2.939599020166669e-06, | |
| "min": 2.939599020166669e-06, | |
| "max": 0.0002969568010144, | |
| "count": 50 | |
| }, | |
| "Pyramids.Policy.LearningRate.sum": { | |
| "value": 2.645639118150002e-05, | |
| "min": 2.645639118150002e-05, | |
| "max": 0.0022392214535929, | |
| "count": 50 | |
| }, | |
| "Pyramids.Policy.Epsilon.mean": { | |
| "value": 0.10097983333333334, | |
| "min": 0.10097983333333334, | |
| "max": 0.19898559999999998, | |
| "count": 50 | |
| }, | |
| "Pyramids.Policy.Epsilon.sum": { | |
| "value": 0.9088185000000001, | |
| "min": 0.7959423999999999, | |
| "max": 1.7299627, | |
| "count": 50 | |
| }, | |
| "Pyramids.Policy.Beta.mean": { | |
| "value": 0.0001078853500000001, | |
| "min": 0.0001078853500000001, | |
| "max": 0.00989866144, | |
| "count": 50 | |
| }, | |
| "Pyramids.Policy.Beta.sum": { | |
| "value": 0.0009709681500000008, | |
| "min": 0.0009709681500000008, | |
| "max": 0.07465606929, | |
| "count": 50 | |
| }, | |
| "Pyramids.Losses.RNDLoss.mean": { | |
| "value": 0.006949213799089193, | |
| "min": 0.006566396448761225, | |
| "max": 0.38224899768829346, | |
| "count": 50 | |
| }, | |
| "Pyramids.Losses.RNDLoss.sum": { | |
| "value": 0.06254292279481888, | |
| "min": 0.060324959456920624, | |
| "max": 1.5289959907531738, | |
| "count": 50 | |
| }, | |
| "Pyramids.Environment.EpisodeLength.mean": { | |
| "value": 374.8474576271187, | |
| "min": 348.4807692307692, | |
| "max": 999.0, | |
| "count": 50 | |
| }, | |
| "Pyramids.Environment.EpisodeLength.sum": { | |
| "value": 22116.0, | |
| "min": 15323.0, | |
| "max": 31968.0, | |
| "count": 50 | |
| }, | |
| "Pyramids.Environment.CumulativeReward.mean": { | |
| "value": 1.6222586012605964, | |
| "min": -1.0000000521540642, | |
| "max": 1.6222586012605964, | |
| "count": 50 | |
| }, | |
| "Pyramids.Environment.CumulativeReward.sum": { | |
| "value": 94.09099887311459, | |
| "min": -32.000001668930054, | |
| "max": 94.09099887311459, | |
| "count": 50 | |
| }, | |
| "Pyramids.Policy.ExtrinsicReward.mean": { | |
| "value": 1.6222586012605964, | |
| "min": -1.0000000521540642, | |
| "max": 1.6222586012605964, | |
| "count": 50 | |
| }, | |
| "Pyramids.Policy.ExtrinsicReward.sum": { | |
| "value": 94.09099887311459, | |
| "min": -32.000001668930054, | |
| "max": 94.09099887311459, | |
| "count": 50 | |
| }, | |
| "Pyramids.Policy.RndReward.mean": { | |
| "value": 0.026957215070006927, | |
| "min": 0.026129583753052948, | |
| "max": 5.808251433540136, | |
| "count": 50 | |
| }, | |
| "Pyramids.Policy.RndReward.sum": { | |
| "value": 1.5635184740604018, | |
| "min": 1.307538619497791, | |
| "max": 92.93202293664217, | |
| "count": 50 | |
| }, | |
| "Pyramids.IsTraining.mean": { | |
| "value": 1.0, | |
| "min": 1.0, | |
| "max": 1.0, | |
| "count": 50 | |
| }, | |
| "Pyramids.IsTraining.sum": { | |
| "value": 1.0, | |
| "min": 1.0, | |
| "max": 1.0, | |
| "count": 50 | |
| } | |
| }, | |
| "metadata": { | |
| "timer_format_version": "0.1.0", | |
| "start_time_seconds": "1787394184", | |
| "python_version": "3.10.12 (main, Jul 26 2023, 13:20:36) [Clang 16.0.3 ]", | |
| "command_line_arguments": ".venv-mlagents/bin/mlagents-learn config/ppo/PyramidsRND.yaml --env=training-envs-executables/linux/Pyramids/Pyramids --run-id=Pyramids1 --no-graphics --force", | |
| "mlagents_version": "1.2.0.dev0", | |
| "mlagents_envs_version": "1.2.0.dev0", | |
| "communication_protocol_version": "1.5.0", | |
| "pytorch_version": "2.8.0+cu128", | |
| "numpy_version": "1.23.5", | |
| "end_time_seconds": "1787395087" | |
| }, | |
| "total": 903.479309686, | |
| "count": 1, | |
| "self": 0.5869812669999419, | |
| "children": { | |
| "run_training.setup": { | |
| "total": 0.0565203360000055, | |
| "count": 1, | |
| "self": 0.0565203360000055 | |
| }, | |
| "TrainerController.start_learning": { | |
| "total": 902.8358080830001, | |
| "count": 1, | |
| "self": 0.7708607759798269, | |
| "children": { | |
| "TrainerController._reset_env": { | |
| "total": 12.615482134000047, | |
| "count": 1, | |
| "self": 12.615482134000047 | |
| }, | |
| "TrainerController.advance": { | |
| "total": 889.3699326650202, | |
| "count": 63652, | |
| "self": 0.7434145150274389, | |
| "children": { | |
| "env_step": { | |
| "total": 536.1517973169932, | |
| "count": 63652, | |
| "self": 449.9551739180137, | |
| "children": { | |
| "SubprocessEnvManager._take_step": { | |
| "total": 85.70504401097747, | |
| "count": 63652, | |
| "self": 2.25659100996387, | |
| "children": { | |
| "TorchPolicy.evaluate": { | |
| "total": 83.4484530010136, | |
| "count": 62559, | |
| "self": 83.4484530010136 | |
| } | |
| } | |
| }, | |
| "workers": { | |
| "total": 0.4915793880020374, | |
| "count": 63652, | |
| "self": 0.0, | |
| "children": { | |
| "worker_root": { | |
| "total": 901.7233150899893, | |
| "count": 63652, | |
| "is_parallel": true, | |
| "self": 502.4046308449831, | |
| "children": { | |
| "run_training.setup": { | |
| "total": 0.0, | |
| "count": 0, | |
| "is_parallel": true, | |
| "self": 0.0, | |
| "children": { | |
| "steps_from_proto": { | |
| "total": 0.012606593999976212, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0021886170000016136, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 0.010417976999974599, | |
| "count": 8, | |
| "is_parallel": true, | |
| "self": 0.010417976999974599 | |
| } | |
| } | |
| }, | |
| "UnityEnvironment.step": { | |
| "total": 0.031868248999955995, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.00022133699997084477, | |
| "children": { | |
| "UnityEnvironment._generate_step_input": { | |
| "total": 0.00022066799999720388, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.00022066799999720388 | |
| }, | |
| "communicator.exchange": { | |
| "total": 0.03067702699996744, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.03067702699996744 | |
| }, | |
| "steps_from_proto": { | |
| "total": 0.0007492170000205078, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0001665239999510959, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 0.0005826930000694119, | |
| "count": 8, | |
| "is_parallel": true, | |
| "self": 0.0005826930000694119 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "UnityEnvironment.step": { | |
| "total": 399.31868424500624, | |
| "count": 63651, | |
| "is_parallel": true, | |
| "self": 11.94500670706384, | |
| "children": { | |
| "UnityEnvironment._generate_step_input": { | |
| "total": 9.272525469963682, | |
| "count": 63651, | |
| "is_parallel": true, | |
| "self": 9.272525469963682 | |
| }, | |
| "communicator.exchange": { | |
| "total": 343.31836329697194, | |
| "count": 63651, | |
| "is_parallel": true, | |
| "self": 343.31836329697194 | |
| }, | |
| "steps_from_proto": { | |
| "total": 34.78278877100678, | |
| "count": 63651, | |
| "is_parallel": true, | |
| "self": 8.213779039975861, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 26.569009731030917, | |
| "count": 509208, | |
| "is_parallel": true, | |
| "self": 26.569009731030917 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "trainer_advance": { | |
| "total": 352.47472083299954, | |
| "count": 63652, | |
| "self": 1.356177588014532, | |
| "children": { | |
| "process_trajectory": { | |
| "total": 64.77816243298406, | |
| "count": 63652, | |
| "self": 62.96707069398411, | |
| "children": { | |
| "RLTrainer._checkpoint": { | |
| "total": 1.8110917389999486, | |
| "count": 10, | |
| "self": 1.8110917389999486 | |
| } | |
| } | |
| }, | |
| "_update_policy": { | |
| "total": 286.34038081200094, | |
| "count": 447, | |
| "self": 153.11913995300836, | |
| "children": { | |
| "TorchPPOOptimizer.update": { | |
| "total": 133.2212408589926, | |
| "count": 22818, | |
| "self": 133.2212408589926 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "trainer_threads": { | |
| "total": 7.010000899754232e-07, | |
| "count": 1, | |
| "self": 7.010000899754232e-07 | |
| }, | |
| "TrainerController._save_models": { | |
| "total": 0.07953180699996665, | |
| "count": 1, | |
| "self": 0.0005698600000414444, | |
| "children": { | |
| "RLTrainer._checkpoint": { | |
| "total": 0.0789619469999252, | |
| "count": 1, | |
| "self": 0.0789619469999252 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } |