Reinforcement Learning
ml-agents
TensorBoard
ONNX
Pyramids
deep-reinforcement-learning
ML-Agents-Pyramids
Instructions to use herurg/ppo-Pyramids with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- ml-agents
How to use herurg/ppo-Pyramids with ml-agents:
mlagents-load-from-hf --repo-id="herurg/ppo-Pyramids" --local-dir="./download: string[]s"
- Notebooks
- Google Colab
- Kaggle
| { | |
| "name": "root", | |
| "gauges": { | |
| "Pyramids.Policy.Entropy.mean": { | |
| "value": 0.2955581843852997, | |
| "min": 0.2747502326965332, | |
| "max": 1.6005821228027344, | |
| "count": 100 | |
| }, | |
| "Pyramids.Policy.Entropy.sum": { | |
| "value": 2998.14208984375, | |
| "min": 2798.183349609375, | |
| "max": 16389.9609375, | |
| "count": 100 | |
| }, | |
| "Pyramids.Step.mean": { | |
| "value": 999951.0, | |
| "min": 9984.0, | |
| "max": 999951.0, | |
| "count": 100 | |
| }, | |
| "Pyramids.Step.sum": { | |
| "value": 999951.0, | |
| "min": 9984.0, | |
| "max": 999951.0, | |
| "count": 100 | |
| }, | |
| "Pyramids.Policy.ExtrinsicValueEstimate.mean": { | |
| "value": 0.6654378175735474, | |
| "min": -0.1540381908416748, | |
| "max": 0.7813151478767395, | |
| "count": 100 | |
| }, | |
| "Pyramids.Policy.ExtrinsicValueEstimate.sum": { | |
| "value": 64.54747009277344, | |
| "min": -12.477093696594238, | |
| "max": 76.56888580322266, | |
| "count": 100 | |
| }, | |
| "Pyramids.Policy.RndValueEstimate.mean": { | |
| "value": 0.005953073501586914, | |
| "min": -0.09729652851819992, | |
| "max": 0.30549734830856323, | |
| "count": 100 | |
| }, | |
| "Pyramids.Policy.RndValueEstimate.sum": { | |
| "value": 0.5774481296539307, | |
| "min": -9.048577308654785, | |
| "max": 24.745285034179688, | |
| "count": 100 | |
| }, | |
| "Pyramids.Losses.PolicyLoss.mean": { | |
| "value": 0.07172110477404203, | |
| "min": 0.0586659834586808, | |
| "max": 0.07878304312765193, | |
| "count": 100 | |
| }, | |
| "Pyramids.Losses.PolicyLoss.sum": { | |
| "value": 0.28688441909616813, | |
| "min": 0.1337058456556406, | |
| "max": 0.39127714224741794, | |
| "count": 100 | |
| }, | |
| "Pyramids.Losses.ValueLoss.mean": { | |
| "value": 0.014707582364887156, | |
| "min": 0.00019025595812889747, | |
| "max": 0.019858406014585248, | |
| "count": 100 | |
| }, | |
| "Pyramids.Losses.ValueLoss.sum": { | |
| "value": 0.05883032945954862, | |
| "min": 0.0007610238325155899, | |
| "max": 0.09929203007292624, | |
| "count": 100 | |
| }, | |
| "Pyramids.Policy.LearningRate.mean": { | |
| "value": 1.4835995054999933e-06, | |
| "min": 1.4835995054999933e-06, | |
| "max": 0.0002981568006144, | |
| "count": 100 | |
| }, | |
| "Pyramids.Policy.LearningRate.sum": { | |
| "value": 5.934398021999973e-06, | |
| "min": 5.934398021999973e-06, | |
| "max": 0.0012819978726674, | |
| "count": 100 | |
| }, | |
| "Pyramids.Policy.Epsilon.mean": { | |
| "value": 0.1004945, | |
| "min": 0.1004945, | |
| "max": 0.1993856, | |
| "count": 100 | |
| }, | |
| "Pyramids.Policy.Epsilon.sum": { | |
| "value": 0.401978, | |
| "min": 0.39717119999999995, | |
| "max": 0.9273326000000001, | |
| "count": 100 | |
| }, | |
| "Pyramids.Policy.Beta.mean": { | |
| "value": 5.940054999999979e-05, | |
| "min": 5.940054999999979e-05, | |
| "max": 0.00993862144, | |
| "count": 100 | |
| }, | |
| "Pyramids.Policy.Beta.sum": { | |
| "value": 0.00023760219999999915, | |
| "min": 0.00023760219999999915, | |
| "max": 0.04274052674, | |
| "count": 100 | |
| }, | |
| "Pyramids.Losses.RNDLoss.mean": { | |
| "value": 0.010158480145037174, | |
| "min": 0.010158480145037174, | |
| "max": 0.6577078700065613, | |
| "count": 100 | |
| }, | |
| "Pyramids.Losses.RNDLoss.sum": { | |
| "value": 0.0406339205801487, | |
| "min": 0.0406339205801487, | |
| "max": 1.3154157400131226, | |
| "count": 100 | |
| }, | |
| "Pyramids.IsTraining.mean": { | |
| "value": 1.0, | |
| "min": 1.0, | |
| "max": 1.0, | |
| "count": 100 | |
| }, | |
| "Pyramids.IsTraining.sum": { | |
| "value": 1.0, | |
| "min": 1.0, | |
| "max": 1.0, | |
| "count": 100 | |
| }, | |
| "Pyramids.Environment.EpisodeLength.mean": { | |
| "value": 257.2631578947368, | |
| "min": 222.47826086956522, | |
| "max": 999.0, | |
| "count": 99 | |
| }, | |
| "Pyramids.Environment.EpisodeLength.sum": { | |
| "value": 9776.0, | |
| "min": 603.0, | |
| "max": 15984.0, | |
| "count": 99 | |
| }, | |
| "Pyramids.Environment.CumulativeReward.mean": { | |
| "value": 1.637436831271962, | |
| "min": -1.0000000521540642, | |
| "max": 1.7800888799958758, | |
| "count": 99 | |
| }, | |
| "Pyramids.Environment.CumulativeReward.sum": { | |
| "value": 62.22259958833456, | |
| "min": -16.000000834465027, | |
| "max": 80.10399959981441, | |
| "count": 99 | |
| }, | |
| "Pyramids.Policy.ExtrinsicReward.mean": { | |
| "value": 1.637436831271962, | |
| "min": -1.0000000521540642, | |
| "max": 1.7800888799958758, | |
| "count": 99 | |
| }, | |
| "Pyramids.Policy.ExtrinsicReward.sum": { | |
| "value": 62.22259958833456, | |
| "min": -16.000000834465027, | |
| "max": 80.10399959981441, | |
| "count": 99 | |
| }, | |
| "Pyramids.Policy.RndReward.mean": { | |
| "value": 0.02774641904933991, | |
| "min": 0.023877781706525842, | |
| "max": 7.177274631336331, | |
| "count": 99 | |
| }, | |
| "Pyramids.Policy.RndReward.sum": { | |
| "value": 1.0543639238749165, | |
| "min": 0.9110401947691571, | |
| "max": 114.8363941013813, | |
| "count": 99 | |
| } | |
| }, | |
| "metadata": { | |
| "timer_format_version": "0.1.0", | |
| "start_time_seconds": "1787486076", | |
| "python_version": "3.10.12 | packaged by conda-forge | (main, Jun 23 2023, 22:40:32) [GCC 12.3.0]", | |
| "command_line_arguments": "/content/mlagents310/bin/mlagents-learn /content/ml-agents/config/ppo/PyramidsRND_HF.yaml --env=/content/ml-agents/training-envs-executables/linux/Pyramids/Pyramids --run-id=Pyramids1 --results-dir=/content/drive/MyDrive/HF_DeepRL_Unit5/Pyramids/results --no-graphics", | |
| "mlagents_version": "1.1.0", | |
| "mlagents_envs_version": "1.1.0", | |
| "communication_protocol_version": "1.5.0", | |
| "pytorch_version": "2.1.1+cpu", | |
| "numpy_version": "1.23.5", | |
| "end_time_seconds": "1787490264" | |
| }, | |
| "total": 4188.032266593002, | |
| "count": 1, | |
| "self": 0.6529515950005589, | |
| "children": { | |
| "run_training.setup": { | |
| "total": 0.03629553599967039, | |
| "count": 1, | |
| "self": 0.03629553599967039 | |
| }, | |
| "TrainerController.start_learning": { | |
| "total": 4187.343019462001, | |
| "count": 1, | |
| "self": 2.094446508881447, | |
| "children": { | |
| "TrainerController._reset_env": { | |
| "total": 1.929896285999348, | |
| "count": 1, | |
| "self": 1.929896285999348 | |
| }, | |
| "TrainerController.advance": { | |
| "total": 4183.104363916122, | |
| "count": 64345, | |
| "self": 2.225138767460521, | |
| "children": { | |
| "env_step": { | |
| "total": 3097.410105198056, | |
| "count": 64345, | |
| "self": 2952.785838075484, | |
| "children": { | |
| "SubprocessEnvManager._take_step": { | |
| "total": 143.23881447907843, | |
| "count": 64345, | |
| "self": 6.317051528889351, | |
| "children": { | |
| "TorchPolicy.evaluate": { | |
| "total": 136.92176295018908, | |
| "count": 62560, | |
| "self": 136.92176295018908 | |
| } | |
| } | |
| }, | |
| "workers": { | |
| "total": 1.3854526434934087, | |
| "count": 64345, | |
| "self": 0.0, | |
| "children": { | |
| "worker_root": { | |
| "total": 4177.959397364599, | |
| "count": 64345, | |
| "is_parallel": true, | |
| "self": 1408.2953628406867, | |
| "children": { | |
| "run_training.setup": { | |
| "total": 0.0, | |
| "count": 0, | |
| "is_parallel": true, | |
| "self": 0.0, | |
| "children": { | |
| "steps_from_proto": { | |
| "total": 0.003124605000266456, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.000906355004190118, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 0.002218249996076338, | |
| "count": 8, | |
| "is_parallel": true, | |
| "self": 0.002218249996076338 | |
| } | |
| } | |
| }, | |
| "UnityEnvironment.step": { | |
| "total": 0.07438645799993537, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0017527049985801568, | |
| "children": { | |
| "UnityEnvironment._generate_step_input": { | |
| "total": 0.0005888550003874116, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0005888550003874116 | |
| }, | |
| "communicator.exchange": { | |
| "total": 0.06870571299987205, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.06870571299987205 | |
| }, | |
| "steps_from_proto": { | |
| "total": 0.003339185001095757, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.00046066399772826117, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 0.0028785210033674957, | |
| "count": 8, | |
| "is_parallel": true, | |
| "self": 0.0028785210033674957 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "UnityEnvironment.step": { | |
| "total": 2769.664034523912, | |
| "count": 64344, | |
| "is_parallel": true, | |
| "self": 51.98580021241287, | |
| "children": { | |
| "UnityEnvironment._generate_step_input": { | |
| "total": 33.44262606215307, | |
| "count": 64344, | |
| "is_parallel": true, | |
| "self": 33.44262606215307 | |
| }, | |
| "communicator.exchange": { | |
| "total": 2521.4422236939354, | |
| "count": 64344, | |
| "is_parallel": true, | |
| "self": 2521.4422236939354 | |
| }, | |
| "steps_from_proto": { | |
| "total": 162.79338455541074, | |
| "count": 64344, | |
| "is_parallel": true, | |
| "self": 32.41883957612845, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 130.3745449792823, | |
| "count": 514752, | |
| "is_parallel": true, | |
| "self": 130.3745449792823 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "trainer_advance": { | |
| "total": 1083.469119950605, | |
| "count": 64345, | |
| "self": 3.7547200758344843, | |
| "children": { | |
| "process_trajectory": { | |
| "total": 160.8547617767672, | |
| "count": 64345, | |
| "self": 159.13201149576344, | |
| "children": { | |
| "RLTrainer._checkpoint": { | |
| "total": 1.722750281003755, | |
| "count": 10, | |
| "self": 1.722750281003755 | |
| } | |
| } | |
| }, | |
| "_update_policy": { | |
| "total": 918.8596380980034, | |
| "count": 452, | |
| "self": 371.8709007099733, | |
| "children": { | |
| "TorchPPOOptimizer.update": { | |
| "total": 546.9887373880301, | |
| "count": 22776, | |
| "self": 546.9887373880301 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "trainer_threads": { | |
| "total": 1.395997969666496e-06, | |
| "count": 1, | |
| "self": 1.395997969666496e-06 | |
| }, | |
| "TrainerController._save_models": { | |
| "total": 0.21431135500097298, | |
| "count": 1, | |
| "self": 0.022026874001312535, | |
| "children": { | |
| "RLTrainer._checkpoint": { | |
| "total": 0.19228448099966045, | |
| "count": 1, | |
| "self": 0.19228448099966045 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } |