Reinforcement Learning
ml-agents
TensorBoard
ONNX
Pyramids
deep-reinforcement-learning
ML-Agents-Pyramids
Instructions to use Kathapult/ppo-Pyramid with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- ml-agents
How to use Kathapult/ppo-Pyramid with ml-agents:
mlagents-load-from-hf --repo-id="Kathapult/ppo-Pyramid" --local-dir="./download: string[]s"
- Notebooks
- Google Colab
- Kaggle
| { | |
| "name": "root", | |
| "gauges": { | |
| "Pyramids.Policy.Entropy.mean": { | |
| "value": 0.5421552062034607, | |
| "min": 0.5109202861785889, | |
| "max": 1.4702211618423462, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Entropy.sum": { | |
| "value": 16351.400390625, | |
| "min": 15384.83203125, | |
| "max": 44600.62890625, | |
| "count": 33 | |
| }, | |
| "Pyramids.Step.mean": { | |
| "value": 989931.0, | |
| "min": 29952.0, | |
| "max": 989931.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Step.sum": { | |
| "value": 989931.0, | |
| "min": 29952.0, | |
| "max": 989931.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicValueEstimate.mean": { | |
| "value": 0.1463415026664734, | |
| "min": -0.09558987617492676, | |
| "max": 0.1810888797044754, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicValueEstimate.sum": { | |
| "value": 36.43903350830078, | |
| "min": -23.037160873413086, | |
| "max": 45.99657440185547, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndValueEstimate.mean": { | |
| "value": 0.005960748065263033, | |
| "min": -0.012978869490325451, | |
| "max": 0.6682460904121399, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndValueEstimate.sum": { | |
| "value": 1.4842262268066406, | |
| "min": -3.296632766723633, | |
| "max": 158.37432861328125, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.PolicyLoss.mean": { | |
| "value": 0.06806919401503723, | |
| "min": 0.06538379723635478, | |
| "max": 0.07575467328867416, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.PolicyLoss.sum": { | |
| "value": 0.9529687162105213, | |
| "min": 0.5060772345807463, | |
| "max": 1.054861680061246, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.ValueLoss.mean": { | |
| "value": 0.009048821650318204, | |
| "min": 0.00015163301405559531, | |
| "max": 0.011161741889211926, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.ValueLoss.sum": { | |
| "value": 0.12668350310445486, | |
| "min": 0.001971229182722739, | |
| "max": 0.15626438644896695, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.LearningRate.mean": { | |
| "value": 7.774704551321427e-06, | |
| "min": 7.774704551321427e-06, | |
| "max": 0.00029515063018788575, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.LearningRate.sum": { | |
| "value": 0.00010884586371849998, | |
| "min": 0.00010884586371849998, | |
| "max": 0.0036333313888895994, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Epsilon.mean": { | |
| "value": 0.10259153571428571, | |
| "min": 0.10259153571428571, | |
| "max": 0.19838354285714285, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Epsilon.sum": { | |
| "value": 1.4362815, | |
| "min": 1.3886848, | |
| "max": 2.611110400000001, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Beta.mean": { | |
| "value": 0.0002688944178571428, | |
| "min": 0.0002688944178571428, | |
| "max": 0.00983851593142857, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Beta.sum": { | |
| "value": 0.0037645218499999995, | |
| "min": 0.0037645218499999995, | |
| "max": 0.12112992896000001, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.RNDLoss.mean": { | |
| "value": 0.0168222077190876, | |
| "min": 0.015354647301137447, | |
| "max": 0.5618124604225159, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.RNDLoss.sum": { | |
| "value": 0.2355109006166458, | |
| "min": 0.2149650603532791, | |
| "max": 3.932687282562256, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.EpisodeLength.mean": { | |
| "value": 704.8048780487804, | |
| "min": 637.1086956521739, | |
| "max": 999.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.EpisodeLength.sum": { | |
| "value": 28897.0, | |
| "min": 15984.0, | |
| "max": 32751.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.CumulativeReward.mean": { | |
| "value": 0.6608926519388105, | |
| "min": -1.0000000521540642, | |
| "max": 0.9040166370881101, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.CumulativeReward.sum": { | |
| "value": 27.096598729491234, | |
| "min": -31.99320164322853, | |
| "max": 43.39279858022928, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicReward.mean": { | |
| "value": 0.6608926519388105, | |
| "min": -1.0000000521540642, | |
| "max": 0.9040166370881101, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicReward.sum": { | |
| "value": 27.096598729491234, | |
| "min": -31.99320164322853, | |
| "max": 43.39279858022928, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndReward.mean": { | |
| "value": 0.12326012659397703, | |
| "min": 0.1069977554216166, | |
| "max": 12.098094806075096, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndReward.sum": { | |
| "value": 5.053665190353058, | |
| "min": 4.975830416660756, | |
| "max": 193.56951689720154, | |
| "count": 33 | |
| }, | |
| "Pyramids.IsTraining.mean": { | |
| "value": 1.0, | |
| "min": 1.0, | |
| "max": 1.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.IsTraining.sum": { | |
| "value": 1.0, | |
| "min": 1.0, | |
| "max": 1.0, | |
| "count": 33 | |
| } | |
| }, | |
| "metadata": { | |
| "timer_format_version": "0.1.0", | |
| "start_time_seconds": "1781878676", | |
| "python_version": "3.10.11 (main, May 16 2023, 00:28:57) [GCC 11.2.0]", | |
| "command_line_arguments": "/usr/local/bin/mlagents-learn ./config/ppo/PyramidsRND.yaml --env=./training-envs-executables/linux/Pyramids/Pyramids --run-id=Pyramids Training --no-graphics", | |
| "mlagents_version": "1.2.0.dev0", | |
| "mlagents_envs_version": "1.2.0.dev0", | |
| "communication_protocol_version": "1.5.0", | |
| "pytorch_version": "2.8.0+cu128", | |
| "numpy_version": "1.23.5", | |
| "end_time_seconds": "1781881029" | |
| }, | |
| "total": 2353.6250749910005, | |
| "count": 1, | |
| "self": 0.5344704550002461, | |
| "children": { | |
| "run_training.setup": { | |
| "total": 0.02381314400008705, | |
| "count": 1, | |
| "self": 0.02381314400008705 | |
| }, | |
| "TrainerController.start_learning": { | |
| "total": 2353.066791392, | |
| "count": 1, | |
| "self": 1.4768955798440402, | |
| "children": { | |
| "TrainerController._reset_env": { | |
| "total": 2.216733632999876, | |
| "count": 1, | |
| "self": 2.216733632999876 | |
| }, | |
| "TrainerController.advance": { | |
| "total": 2349.286795397156, | |
| "count": 63365, | |
| "self": 1.450732352187515, | |
| "children": { | |
| "env_step": { | |
| "total": 1698.5999840099662, | |
| "count": 63365, | |
| "self": 1536.7075065429908, | |
| "children": { | |
| "SubprocessEnvManager._take_step": { | |
| "total": 161.0474582819529, | |
| "count": 63365, | |
| "self": 4.891472954865094, | |
| "children": { | |
| "TorchPolicy.evaluate": { | |
| "total": 156.15598532708782, | |
| "count": 62563, | |
| "self": 156.15598532708782 | |
| } | |
| } | |
| }, | |
| "workers": { | |
| "total": 0.8450191850224655, | |
| "count": 63365, | |
| "self": 0.0, | |
| "children": { | |
| "worker_root": { | |
| "total": 2347.404555083034, | |
| "count": 63365, | |
| "is_parallel": true, | |
| "self": 933.9190591480174, | |
| "children": { | |
| "run_training.setup": { | |
| "total": 0.0, | |
| "count": 0, | |
| "is_parallel": true, | |
| "self": 0.0, | |
| "children": { | |
| "steps_from_proto": { | |
| "total": 0.0019775870000557916, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.000693468999088509, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 0.0012841180009672826, | |
| "count": 8, | |
| "is_parallel": true, | |
| "self": 0.0012841180009672826 | |
| } | |
| } | |
| }, | |
| "UnityEnvironment.step": { | |
| "total": 0.05364939500032051, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0005865680004717433, | |
| "children": { | |
| "UnityEnvironment._generate_step_input": { | |
| "total": 0.0004667029998017824, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0004667029998017824 | |
| }, | |
| "communicator.exchange": { | |
| "total": 0.0508433420000074, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0508433420000074 | |
| }, | |
| "steps_from_proto": { | |
| "total": 0.0017527820000395877, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.00041325099982714164, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 0.001339531000212446, | |
| "count": 8, | |
| "is_parallel": true, | |
| "self": 0.001339531000212446 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "UnityEnvironment.step": { | |
| "total": 1413.4854959350168, | |
| "count": 63364, | |
| "is_parallel": true, | |
| "self": 35.20592063989125, | |
| "children": { | |
| "UnityEnvironment._generate_step_input": { | |
| "total": 25.055125791067894, | |
| "count": 63364, | |
| "is_parallel": true, | |
| "self": 25.055125791067894 | |
| }, | |
| "communicator.exchange": { | |
| "total": 1238.0361629739941, | |
| "count": 63364, | |
| "is_parallel": true, | |
| "self": 1238.0361629739941 | |
| }, | |
| "steps_from_proto": { | |
| "total": 115.18828653006358, | |
| "count": 63364, | |
| "is_parallel": true, | |
| "self": 23.549093686811375, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 91.6391928432522, | |
| "count": 506912, | |
| "is_parallel": true, | |
| "self": 91.6391928432522 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "trainer_advance": { | |
| "total": 649.2360790350021, | |
| "count": 63365, | |
| "self": 2.729695350950351, | |
| "children": { | |
| "process_trajectory": { | |
| "total": 115.21554636805104, | |
| "count": 63365, | |
| "self": 114.96722944405155, | |
| "children": { | |
| "RLTrainer._checkpoint": { | |
| "total": 0.24831692399948224, | |
| "count": 2, | |
| "self": 0.24831692399948224 | |
| } | |
| } | |
| }, | |
| "_update_policy": { | |
| "total": 531.2908373160008, | |
| "count": 450, | |
| "self": 282.17515017301776, | |
| "children": { | |
| "TorchPPOOptimizer.update": { | |
| "total": 249.115687142983, | |
| "count": 22764, | |
| "self": 249.115687142983 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "trainer_threads": { | |
| "total": 8.510005500284024e-07, | |
| "count": 1, | |
| "self": 8.510005500284024e-07 | |
| }, | |
| "TrainerController._save_models": { | |
| "total": 0.08636593099981837, | |
| "count": 1, | |
| "self": 0.0010633989995767479, | |
| "children": { | |
| "RLTrainer._checkpoint": { | |
| "total": 0.08530253200024163, | |
| "count": 1, | |
| "self": 0.08530253200024163 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } |