Reinforcement Learning
ml-agents
TensorBoard
ONNX
Pyramids
deep-reinforcement-learning
ML-Agents-Pyramids
Instructions to use fklska/ppo-PyramidsRND with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- ml-agents
How to use fklska/ppo-PyramidsRND with ml-agents:
mlagents-load-from-hf --repo-id="fklska/ppo-PyramidsRND" --local-dir="./download: string[]s"
- Notebooks
- Google Colab
- Kaggle
| { | |
| "name": "root", | |
| "gauges": { | |
| "Pyramids.Policy.Entropy.mean": { | |
| "value": 0.3501512110233307, | |
| "min": 0.33211055397987366, | |
| "max": 1.3325566053390503, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Entropy.sum": { | |
| "value": 10605.3798828125, | |
| "min": 9969.12890625, | |
| "max": 40424.4375, | |
| "count": 33 | |
| }, | |
| "Pyramids.Step.mean": { | |
| "value": 989905.0, | |
| "min": 29952.0, | |
| "max": 989905.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Step.sum": { | |
| "value": 989905.0, | |
| "min": 29952.0, | |
| "max": 989905.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicValueEstimate.mean": { | |
| "value": 0.7288305759429932, | |
| "min": -0.07709911465644836, | |
| "max": 0.740848183631897, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicValueEstimate.sum": { | |
| "value": 211.36087036132812, | |
| "min": -18.65798568725586, | |
| "max": 212.62342834472656, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndValueEstimate.mean": { | |
| "value": -0.009867994114756584, | |
| "min": -0.011822749860584736, | |
| "max": 0.40765824913978577, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndValueEstimate.sum": { | |
| "value": -2.8617184162139893, | |
| "min": -2.9911556243896484, | |
| "max": 96.61500549316406, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.PolicyLoss.mean": { | |
| "value": 0.0688488850417863, | |
| "min": 0.0640230809267248, | |
| "max": 0.0739196422928879, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.PolicyLoss.sum": { | |
| "value": 0.9638843905850081, | |
| "min": 0.5004837631919332, | |
| "max": 1.0650410116159394, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.ValueLoss.mean": { | |
| "value": 0.014359603087844638, | |
| "min": 0.0014472519972324755, | |
| "max": 0.016265684270322146, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.ValueLoss.sum": { | |
| "value": 0.20103444322982494, | |
| "min": 0.010130763980627329, | |
| "max": 0.22771957978451005, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.LearningRate.mean": { | |
| "value": 7.637861739792855e-06, | |
| "min": 7.637861739792855e-06, | |
| "max": 0.00029515063018788575, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.LearningRate.sum": { | |
| "value": 0.00010693006435709998, | |
| "min": 0.00010693006435709998, | |
| "max": 0.003508976330341299, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Epsilon.mean": { | |
| "value": 0.10254592142857143, | |
| "min": 0.10254592142857143, | |
| "max": 0.19838354285714285, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Epsilon.sum": { | |
| "value": 1.4356429, | |
| "min": 1.3691136000000002, | |
| "max": 2.5696587, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Beta.mean": { | |
| "value": 0.0002643375507142857, | |
| "min": 0.0002643375507142857, | |
| "max": 0.00983851593142857, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Beta.sum": { | |
| "value": 0.00370072571, | |
| "min": 0.00370072571, | |
| "max": 0.11698890412999999, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.RNDLoss.mean": { | |
| "value": 0.008767274208366871, | |
| "min": 0.008767274208366871, | |
| "max": 0.374515563249588, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.RNDLoss.sum": { | |
| "value": 0.12274183332920074, | |
| "min": 0.12274183332920074, | |
| "max": 2.6216089725494385, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.EpisodeLength.mean": { | |
| "value": 262.3719008264463, | |
| "min": 262.3719008264463, | |
| "max": 999.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.EpisodeLength.sum": { | |
| "value": 31747.0, | |
| "min": 15984.0, | |
| "max": 32267.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.CumulativeReward.mean": { | |
| "value": 1.6714826380172052, | |
| "min": -1.0000000521540642, | |
| "max": 1.7351727137511426, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.CumulativeReward.sum": { | |
| "value": 202.24939920008183, | |
| "min": -32.000001668930054, | |
| "max": 202.24939920008183, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicReward.mean": { | |
| "value": 1.6714826380172052, | |
| "min": -1.0000000521540642, | |
| "max": 1.7351727137511426, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicReward.sum": { | |
| "value": 202.24939920008183, | |
| "min": -32.000001668930054, | |
| "max": 202.24939920008183, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndReward.mean": { | |
| "value": 0.023441952148843016, | |
| "min": 0.023441952148843016, | |
| "max": 8.2253372464329, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndReward.sum": { | |
| "value": 2.836476210010005, | |
| "min": 2.6988735643972177, | |
| "max": 131.6053959429264, | |
| "count": 33 | |
| }, | |
| "Pyramids.IsTraining.mean": { | |
| "value": 1.0, | |
| "min": 1.0, | |
| "max": 1.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.IsTraining.sum": { | |
| "value": 1.0, | |
| "min": 1.0, | |
| "max": 1.0, | |
| "count": 33 | |
| } | |
| }, | |
| "metadata": { | |
| "timer_format_version": "0.1.0", | |
| "start_time_seconds": "1784541843", | |
| "python_version": "3.10.12 | packaged by conda-forge | (main, Jun 23 2023, 22:40:32) [GCC 12.3.0]", | |
| "command_line_arguments": "/usr/local/bin/mlagents-learn ./config/ppo/PyramidsRND.yaml --env=./training-envs-executables/linux/Pyramids/Pyramids --run-id=Pyramids Training --no-graphics", | |
| "mlagents_version": "1.2.0.dev0", | |
| "mlagents_envs_version": "1.2.0.dev0", | |
| "communication_protocol_version": "1.5.0", | |
| "pytorch_version": "2.8.0+cu128", | |
| "numpy_version": "1.23.5", | |
| "end_time_seconds": "1784544595" | |
| }, | |
| "total": 2752.782595838, | |
| "count": 1, | |
| "self": 0.528103414000725, | |
| "children": { | |
| "run_training.setup": { | |
| "total": 0.027646086999993713, | |
| "count": 1, | |
| "self": 0.027646086999993713 | |
| }, | |
| "TrainerController.start_learning": { | |
| "total": 2752.226846336999, | |
| "count": 1, | |
| "self": 1.713667019859713, | |
| "children": { | |
| "TrainerController._reset_env": { | |
| "total": 2.247410148999734, | |
| "count": 1, | |
| "self": 2.247410148999734 | |
| }, | |
| "TrainerController.advance": { | |
| "total": 2748.1689283751402, | |
| "count": 64315, | |
| "self": 1.781306609918829, | |
| "children": { | |
| "env_step": { | |
| "total": 2062.6223466560677, | |
| "count": 64315, | |
| "self": 1886.6379189983309, | |
| "children": { | |
| "SubprocessEnvManager._take_step": { | |
| "total": 174.92111731172918, | |
| "count": 64315, | |
| "self": 5.3885907008452705, | |
| "children": { | |
| "TorchPolicy.evaluate": { | |
| "total": 169.5325266108839, | |
| "count": 62549, | |
| "self": 169.5325266108839 | |
| } | |
| } | |
| }, | |
| "workers": { | |
| "total": 1.063310346007711, | |
| "count": 64315, | |
| "self": 0.0, | |
| "children": { | |
| "worker_root": { | |
| "total": 2745.9107308560942, | |
| "count": 64315, | |
| "is_parallel": true, | |
| "self": 998.7776048480855, | |
| "children": { | |
| "run_training.setup": { | |
| "total": 0.0, | |
| "count": 0, | |
| "is_parallel": true, | |
| "self": 0.0, | |
| "children": { | |
| "steps_from_proto": { | |
| "total": 0.0017775019996406627, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0005761600004916545, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 0.0012013419991490082, | |
| "count": 8, | |
| "is_parallel": true, | |
| "self": 0.0012013419991490082 | |
| } | |
| } | |
| }, | |
| "UnityEnvironment.step": { | |
| "total": 0.053902637999271974, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0005990769996060408, | |
| "children": { | |
| "UnityEnvironment._generate_step_input": { | |
| "total": 0.00057317799928569, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.00057317799928569 | |
| }, | |
| "communicator.exchange": { | |
| "total": 0.05094423700029438, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.05094423700029438 | |
| }, | |
| "steps_from_proto": { | |
| "total": 0.0017861460000858642, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.00035048700283368817, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 0.001435658997252176, | |
| "count": 8, | |
| "is_parallel": true, | |
| "self": 0.001435658997252176 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "UnityEnvironment.step": { | |
| "total": 1747.1331260080087, | |
| "count": 64314, | |
| "is_parallel": true, | |
| "self": 36.65351603009822, | |
| "children": { | |
| "UnityEnvironment._generate_step_input": { | |
| "total": 26.03310885281644, | |
| "count": 64314, | |
| "is_parallel": true, | |
| "self": 26.03310885281644 | |
| }, | |
| "communicator.exchange": { | |
| "total": 1560.3493709020531, | |
| "count": 64314, | |
| "is_parallel": true, | |
| "self": 1560.3493709020531 | |
| }, | |
| "steps_from_proto": { | |
| "total": 124.09713022304095, | |
| "count": 64314, | |
| "is_parallel": true, | |
| "self": 25.686523472593763, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 98.41060675044719, | |
| "count": 514512, | |
| "is_parallel": true, | |
| "self": 98.41060675044719 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "trainer_advance": { | |
| "total": 683.7652751091537, | |
| "count": 64315, | |
| "self": 3.3280532641347236, | |
| "children": { | |
| "process_trajectory": { | |
| "total": 125.5387294600232, | |
| "count": 64315, | |
| "self": 125.33758603602382, | |
| "children": { | |
| "RLTrainer._checkpoint": { | |
| "total": 0.20114342399938323, | |
| "count": 2, | |
| "self": 0.20114342399938323 | |
| } | |
| } | |
| }, | |
| "_update_policy": { | |
| "total": 554.8984923849957, | |
| "count": 451, | |
| "self": 298.74812579199715, | |
| "children": { | |
| "TorchPPOOptimizer.update": { | |
| "total": 256.1503665929986, | |
| "count": 22824, | |
| "self": 256.1503665929986 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "trainer_threads": { | |
| "total": 1.077999513654504e-06, | |
| "count": 1, | |
| "self": 1.077999513654504e-06 | |
| }, | |
| "TrainerController._save_models": { | |
| "total": 0.09683971499998734, | |
| "count": 1, | |
| "self": 0.0010708330000852584, | |
| "children": { | |
| "RLTrainer._checkpoint": { | |
| "total": 0.09576888199990208, | |
| "count": 1, | |
| "self": 0.09576888199990208 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } |