Reinforcement Learning
ml-agents
TensorBoard
ONNX
Pyramids
deep-reinforcement-learning
ML-Agents-Pyramids
Instructions to use javiarmijo/PyramidsRND with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- ml-agents
How to use javiarmijo/PyramidsRND with ml-agents:
mlagents-load-from-hf --repo-id="javiarmijo/PyramidsRND" --local-dir="./download: string[]s"
- Notebooks
- Google Colab
- Kaggle
| { | |
| "name": "root", | |
| "gauges": { | |
| "Pyramids.Policy.Entropy.mean": { | |
| "value": 0.4844374358654022, | |
| "min": 0.4844374358654022, | |
| "max": 1.4626991748809814, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Entropy.sum": { | |
| "value": 14548.625, | |
| "min": 14548.625, | |
| "max": 44372.44140625, | |
| "count": 33 | |
| }, | |
| "Pyramids.Step.mean": { | |
| "value": 989922.0, | |
| "min": 29952.0, | |
| "max": 989922.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Step.sum": { | |
| "value": 989922.0, | |
| "min": 29952.0, | |
| "max": 989922.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicValueEstimate.mean": { | |
| "value": 0.5225057005882263, | |
| "min": -0.2971446216106415, | |
| "max": 0.5507718920707703, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicValueEstimate.sum": { | |
| "value": 146.8240966796875, | |
| "min": -70.42327880859375, | |
| "max": 154.76690673828125, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndValueEstimate.mean": { | |
| "value": 0.005472871940582991, | |
| "min": -0.004757067188620567, | |
| "max": 0.3024058938026428, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndValueEstimate.sum": { | |
| "value": 1.5378769636154175, | |
| "min": -1.2844080924987793, | |
| "max": 72.5774154663086, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.PolicyLoss.mean": { | |
| "value": 0.06532027134588654, | |
| "min": 0.0652111707437775, | |
| "max": 0.07324379556701474, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.PolicyLoss.sum": { | |
| "value": 0.9144837988424115, | |
| "min": 0.4916578374494298, | |
| "max": 1.0644322500424916, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.ValueLoss.mean": { | |
| "value": 0.01680323294844822, | |
| "min": 0.00027062356042735806, | |
| "max": 0.01680323294844822, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.ValueLoss.sum": { | |
| "value": 0.23524526127827508, | |
| "min": 0.0029768591647009388, | |
| "max": 0.23524526127827508, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.LearningRate.mean": { | |
| "value": 7.718518855764286e-06, | |
| "min": 7.718518855764286e-06, | |
| "max": 0.00029515063018788575, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.LearningRate.sum": { | |
| "value": 0.0001080592639807, | |
| "min": 0.0001080592639807, | |
| "max": 0.0033797054734316006, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Epsilon.mean": { | |
| "value": 0.10257280714285714, | |
| "min": 0.10257280714285714, | |
| "max": 0.19838354285714285, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Epsilon.sum": { | |
| "value": 1.4360193, | |
| "min": 1.3691136000000002, | |
| "max": 2.5265684, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Beta.mean": { | |
| "value": 0.00026702343357142856, | |
| "min": 0.00026702343357142856, | |
| "max": 0.00983851593142857, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Beta.sum": { | |
| "value": 0.00373832807, | |
| "min": 0.00373832807, | |
| "max": 0.11268418316, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.RNDLoss.mean": { | |
| "value": 0.010689104907214642, | |
| "min": 0.010351241566240788, | |
| "max": 0.406168669462204, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.RNDLoss.sum": { | |
| "value": 0.14964747428894043, | |
| "min": 0.14491738379001617, | |
| "max": 2.8431806564331055, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.EpisodeLength.mean": { | |
| "value": 343.83720930232556, | |
| "min": 343.83720930232556, | |
| "max": 999.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.EpisodeLength.sum": { | |
| "value": 29570.0, | |
| "min": 15984.0, | |
| "max": 32297.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.CumulativeReward.mean": { | |
| "value": 1.5863790496490722, | |
| "min": -1.0000000521540642, | |
| "max": 1.5974092846519725, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.CumulativeReward.sum": { | |
| "value": 136.4285982698202, | |
| "min": -32.000001668930054, | |
| "max": 137.37719848006964, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicReward.mean": { | |
| "value": 1.5863790496490722, | |
| "min": -1.0000000521540642, | |
| "max": 1.5974092846519725, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicReward.sum": { | |
| "value": 136.4285982698202, | |
| "min": -32.000001668930054, | |
| "max": 137.37719848006964, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndReward.mean": { | |
| "value": 0.03778216891331858, | |
| "min": 0.03778216891331858, | |
| "max": 8.018827424384654, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndReward.sum": { | |
| "value": 3.249266526545398, | |
| "min": 3.249266526545398, | |
| "max": 128.30123879015446, | |
| "count": 33 | |
| }, | |
| "Pyramids.IsTraining.mean": { | |
| "value": 1.0, | |
| "min": 1.0, | |
| "max": 1.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.IsTraining.sum": { | |
| "value": 1.0, | |
| "min": 1.0, | |
| "max": 1.0, | |
| "count": 33 | |
| } | |
| }, | |
| "metadata": { | |
| "timer_format_version": "0.1.0", | |
| "start_time_seconds": "1781691731", | |
| "python_version": "3.10.12 (main, Jul 5 2023, 18:54:27) [GCC 11.2.0]", | |
| "command_line_arguments": "/usr/local/bin/mlagents-learn ./config/ppo/PyramidsRND.yaml --env=./training-envs-executables/linux/Pyramids/Pyramids --run-id=Pyramids Training --no-graphics", | |
| "mlagents_version": "1.2.0.dev0", | |
| "mlagents_envs_version": "1.2.0.dev0", | |
| "communication_protocol_version": "1.5.0", | |
| "pytorch_version": "2.8.0+cu128", | |
| "numpy_version": "1.23.5", | |
| "end_time_seconds": "1781694108" | |
| }, | |
| "total": 2376.7685926530003, | |
| "count": 1, | |
| "self": 0.48073778300067715, | |
| "children": { | |
| "run_training.setup": { | |
| "total": 0.03741145599997253, | |
| "count": 1, | |
| "self": 0.03741145599997253 | |
| }, | |
| "TrainerController.start_learning": { | |
| "total": 2376.2504434139996, | |
| "count": 1, | |
| "self": 1.369329333978385, | |
| "children": { | |
| "TrainerController._reset_env": { | |
| "total": 2.652903357000014, | |
| "count": 1, | |
| "self": 2.652903357000014 | |
| }, | |
| "TrainerController.advance": { | |
| "total": 2372.152803797022, | |
| "count": 63734, | |
| "self": 1.4162058960728245, | |
| "children": { | |
| "env_step": { | |
| "total": 1704.6123179129395, | |
| "count": 63734, | |
| "self": 1549.819857362906, | |
| "children": { | |
| "SubprocessEnvManager._take_step": { | |
| "total": 153.97994042697383, | |
| "count": 63734, | |
| "self": 4.706254008973701, | |
| "children": { | |
| "TorchPolicy.evaluate": { | |
| "total": 149.27368641800012, | |
| "count": 62568, | |
| "self": 149.27368641800012 | |
| } | |
| } | |
| }, | |
| "workers": { | |
| "total": 0.8125201230595849, | |
| "count": 63734, | |
| "self": 0.0, | |
| "children": { | |
| "worker_root": { | |
| "total": 2370.4492502349513, | |
| "count": 63734, | |
| "is_parallel": true, | |
| "self": 939.10941228006, | |
| "children": { | |
| "run_training.setup": { | |
| "total": 0.0, | |
| "count": 0, | |
| "is_parallel": true, | |
| "self": 0.0, | |
| "children": { | |
| "steps_from_proto": { | |
| "total": 0.0029095649999817397, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0007659600000806677, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 0.002143604999901072, | |
| "count": 8, | |
| "is_parallel": true, | |
| "self": 0.002143604999901072 | |
| } | |
| } | |
| }, | |
| "UnityEnvironment.step": { | |
| "total": 0.04605088199991769, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0005466000000069471, | |
| "children": { | |
| "UnityEnvironment._generate_step_input": { | |
| "total": 0.000528171999803817, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.000528171999803817 | |
| }, | |
| "communicator.exchange": { | |
| "total": 0.04344047800009321, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.04344047800009321 | |
| }, | |
| "steps_from_proto": { | |
| "total": 0.0015356320000137202, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.00032978299964270263, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 0.0012058490003710176, | |
| "count": 8, | |
| "is_parallel": true, | |
| "self": 0.0012058490003710176 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "UnityEnvironment.step": { | |
| "total": 1431.3398379548912, | |
| "count": 63733, | |
| "is_parallel": true, | |
| "self": 34.11497534489581, | |
| "children": { | |
| "UnityEnvironment._generate_step_input": { | |
| "total": 23.657517906056228, | |
| "count": 63733, | |
| "is_parallel": true, | |
| "self": 23.657517906056228 | |
| }, | |
| "communicator.exchange": { | |
| "total": 1264.2375593099791, | |
| "count": 63733, | |
| "is_parallel": true, | |
| "self": 1264.2375593099791 | |
| }, | |
| "steps_from_proto": { | |
| "total": 109.32978539396004, | |
| "count": 63733, | |
| "is_parallel": true, | |
| "self": 23.015591065302488, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 86.31419432865755, | |
| "count": 509864, | |
| "is_parallel": true, | |
| "self": 86.31419432865755 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "trainer_advance": { | |
| "total": 666.1242799880099, | |
| "count": 63734, | |
| "self": 2.6047069269864096, | |
| "children": { | |
| "process_trajectory": { | |
| "total": 125.10462803202859, | |
| "count": 63734, | |
| "self": 124.89790180302793, | |
| "children": { | |
| "RLTrainer._checkpoint": { | |
| "total": 0.20672622900065107, | |
| "count": 2, | |
| "self": 0.20672622900065107 | |
| } | |
| } | |
| }, | |
| "_update_policy": { | |
| "total": 538.4149450289949, | |
| "count": 447, | |
| "self": 296.48849648000623, | |
| "children": { | |
| "TorchPPOOptimizer.update": { | |
| "total": 241.92644854898867, | |
| "count": 22797, | |
| "self": 241.92644854898867 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "trainer_threads": { | |
| "total": 8.170000000973232e-07, | |
| "count": 1, | |
| "self": 8.170000000973232e-07 | |
| }, | |
| "TrainerController._save_models": { | |
| "total": 0.07540610899923195, | |
| "count": 1, | |
| "self": 0.0012114609990021563, | |
| "children": { | |
| "RLTrainer._checkpoint": { | |
| "total": 0.07419464800022979, | |
| "count": 1, | |
| "self": 0.07419464800022979 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } |