Reinforcement Learning
ml-agents
TensorBoard
ONNX
Pyramids
deep-reinforcement-learning
ML-Agents-Pyramids
Instructions to use rixhi05/ppo-Pyramids with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- ml-agents
How to use rixhi05/ppo-Pyramids with ml-agents:
mlagents-load-from-hf --repo-id="rixhi05/ppo-Pyramids" --local-dir="./download: string[]s"
- Notebooks
- Google Colab
- Kaggle
| { | |
| "name": "root", | |
| "gauges": { | |
| "Pyramids.Policy.Entropy.mean": { | |
| "value": 0.14280743896961212, | |
| "min": 0.13818636536598206, | |
| "max": 1.4633220434188843, | |
| "count": 100 | |
| }, | |
| "Pyramids.Policy.Entropy.sum": { | |
| "value": 4293.36279296875, | |
| "min": 4103.58251953125, | |
| "max": 44391.3359375, | |
| "count": 100 | |
| }, | |
| "Pyramids.Step.mean": { | |
| "value": 2999901.0, | |
| "min": 29952.0, | |
| "max": 2999901.0, | |
| "count": 100 | |
| }, | |
| "Pyramids.Step.sum": { | |
| "value": 2999901.0, | |
| "min": 29952.0, | |
| "max": 2999901.0, | |
| "count": 100 | |
| }, | |
| "Pyramids.Policy.ExtrinsicValueEstimate.mean": { | |
| "value": 0.8156636953353882, | |
| "min": -0.10103955864906311, | |
| "max": 0.9036851525306702, | |
| "count": 100 | |
| }, | |
| "Pyramids.Policy.ExtrinsicValueEstimate.sum": { | |
| "value": 245.5147705078125, | |
| "min": -24.45157241821289, | |
| "max": 280.14239501953125, | |
| "count": 100 | |
| }, | |
| "Pyramids.Policy.RndValueEstimate.mean": { | |
| "value": 0.014001032337546349, | |
| "min": -0.04318656399846077, | |
| "max": 0.36548855900764465, | |
| "count": 100 | |
| }, | |
| "Pyramids.Policy.RndValueEstimate.sum": { | |
| "value": 4.214310646057129, | |
| "min": -10.926200866699219, | |
| "max": 86.62078857421875, | |
| "count": 100 | |
| }, | |
| "Pyramids.Losses.PolicyLoss.mean": { | |
| "value": 0.06845179736791622, | |
| "min": 0.06272790234905071, | |
| "max": 0.07504414065070229, | |
| "count": 100 | |
| }, | |
| "Pyramids.Losses.PolicyLoss.sum": { | |
| "value": 0.958325163150827, | |
| "min": 0.49780014440438697, | |
| "max": 1.0608395482915656, | |
| "count": 100 | |
| }, | |
| "Pyramids.Losses.ValueLoss.mean": { | |
| "value": 0.014363895175241243, | |
| "min": 0.00034970744883078, | |
| "max": 0.01763231658073242, | |
| "count": 100 | |
| }, | |
| "Pyramids.Losses.ValueLoss.sum": { | |
| "value": 0.2010945324533774, | |
| "min": 0.00419648938596936, | |
| "max": 0.24685243213025387, | |
| "count": 100 | |
| }, | |
| "Pyramids.Policy.LearningRate.mean": { | |
| "value": 1.485228076385716e-06, | |
| "min": 1.485228076385716e-06, | |
| "max": 0.00029838354339596195, | |
| "count": 100 | |
| }, | |
| "Pyramids.Policy.LearningRate.sum": { | |
| "value": 2.0793193069400022e-05, | |
| "min": 2.0793193069400022e-05, | |
| "max": 0.003968951577016166, | |
| "count": 100 | |
| }, | |
| "Pyramids.Policy.Epsilon.mean": { | |
| "value": 0.10049504285714285, | |
| "min": 0.10049504285714285, | |
| "max": 0.19946118095238097, | |
| "count": 100 | |
| }, | |
| "Pyramids.Policy.Epsilon.sum": { | |
| "value": 1.4069306, | |
| "min": 1.3962282666666668, | |
| "max": 2.7525246000000005, | |
| "count": 100 | |
| }, | |
| "Pyramids.Policy.Beta.mean": { | |
| "value": 5.9454781428571485e-05, | |
| "min": 5.9454781428571485e-05, | |
| "max": 0.009946171977142856, | |
| "count": 100 | |
| }, | |
| "Pyramids.Policy.Beta.sum": { | |
| "value": 0.0008323669400000008, | |
| "min": 0.0008323669400000008, | |
| "max": 0.13230608494999999, | |
| "count": 100 | |
| }, | |
| "Pyramids.Losses.RNDLoss.mean": { | |
| "value": 0.005520425736904144, | |
| "min": 0.005485298577696085, | |
| "max": 0.5199190378189087, | |
| "count": 100 | |
| }, | |
| "Pyramids.Losses.RNDLoss.sum": { | |
| "value": 0.07728596031665802, | |
| "min": 0.07728596031665802, | |
| "max": 3.6394331455230713, | |
| "count": 100 | |
| }, | |
| "Pyramids.Environment.EpisodeLength.mean": { | |
| "value": 228.34375, | |
| "min": 206.06474820143885, | |
| "max": 999.0, | |
| "count": 100 | |
| }, | |
| "Pyramids.Environment.EpisodeLength.sum": { | |
| "value": 29228.0, | |
| "min": 15984.0, | |
| "max": 33209.0, | |
| "count": 100 | |
| }, | |
| "Pyramids.Environment.CumulativeReward.mean": { | |
| "value": 1.7403952978202142, | |
| "min": -1.0000000521540642, | |
| "max": 1.7939352364205627, | |
| "count": 100 | |
| }, | |
| "Pyramids.Environment.CumulativeReward.sum": { | |
| "value": 222.77059812098742, | |
| "min": -31.998401656746864, | |
| "max": 253.590997710824, | |
| "count": 100 | |
| }, | |
| "Pyramids.Policy.ExtrinsicReward.mean": { | |
| "value": 1.7403952978202142, | |
| "min": -1.0000000521540642, | |
| "max": 1.7939352364205627, | |
| "count": 100 | |
| }, | |
| "Pyramids.Policy.ExtrinsicReward.sum": { | |
| "value": 222.77059812098742, | |
| "min": -31.998401656746864, | |
| "max": 253.590997710824, | |
| "count": 100 | |
| }, | |
| "Pyramids.Policy.RndReward.mean": { | |
| "value": 0.013073411428308646, | |
| "min": 0.012292344707661164, | |
| "max": 10.69173732586205, | |
| "count": 100 | |
| }, | |
| "Pyramids.Policy.RndReward.sum": { | |
| "value": 1.6733966628235066, | |
| "min": 1.6554624012933346, | |
| "max": 171.0677972137928, | |
| "count": 100 | |
| }, | |
| "Pyramids.IsTraining.mean": { | |
| "value": 1.0, | |
| "min": 1.0, | |
| "max": 1.0, | |
| "count": 100 | |
| }, | |
| "Pyramids.IsTraining.sum": { | |
| "value": 1.0, | |
| "min": 1.0, | |
| "max": 1.0, | |
| "count": 100 | |
| } | |
| }, | |
| "metadata": { | |
| "timer_format_version": "0.1.0", | |
| "start_time_seconds": "1785290201", | |
| "python_version": "3.10.12 (main, Jul 5 2023, 18:54:27) [GCC 11.2.0]", | |
| "command_line_arguments": "/usr/local/bin/mlagents-learn ./config/ppo/PyramidsRND.yaml --env=./training-envs-executables/linux/Pyramids/Pyramids --run-id=Pyramids Training --no-graphics", | |
| "mlagents_version": "1.2.0.dev0", | |
| "mlagents_envs_version": "1.2.0.dev0", | |
| "communication_protocol_version": "1.5.0", | |
| "pytorch_version": "2.8.0+cu128", | |
| "numpy_version": "1.23.5", | |
| "end_time_seconds": "1785300165" | |
| }, | |
| "total": 9964.010561595001, | |
| "count": 1, | |
| "self": 0.4790026030023, | |
| "children": { | |
| "run_training.setup": { | |
| "total": 0.02824461499994868, | |
| "count": 1, | |
| "self": 0.02824461499994868 | |
| }, | |
| "TrainerController.start_learning": { | |
| "total": 9963.503314377, | |
| "count": 1, | |
| "self": 5.967677134785845, | |
| "children": { | |
| "TrainerController._reset_env": { | |
| "total": 2.4221490079999057, | |
| "count": 1, | |
| "self": 2.4221490079999057 | |
| }, | |
| "TrainerController.advance": { | |
| "total": 9955.032944610215, | |
| "count": 194776, | |
| "self": 6.214308647095095, | |
| "children": { | |
| "env_step": { | |
| "total": 7667.329358111002, | |
| "count": 194776, | |
| "self": 7053.170723272273, | |
| "children": { | |
| "SubprocessEnvManager._take_step": { | |
| "total": 610.4611874448665, | |
| "count": 194776, | |
| "self": 18.01653528391398, | |
| "children": { | |
| "TorchPolicy.evaluate": { | |
| "total": 592.4446521609525, | |
| "count": 187564, | |
| "self": 592.4446521609525 | |
| } | |
| } | |
| }, | |
| "workers": { | |
| "total": 3.697447393863058, | |
| "count": 194776, | |
| "self": 0.0, | |
| "children": { | |
| "worker_root": { | |
| "total": 9942.324176570171, | |
| "count": 194776, | |
| "is_parallel": true, | |
| "self": 3368.048026751987, | |
| "children": { | |
| "run_training.setup": { | |
| "total": 0.0, | |
| "count": 0, | |
| "is_parallel": true, | |
| "self": 0.0, | |
| "children": { | |
| "steps_from_proto": { | |
| "total": 0.002086856000005355, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0006864360000236047, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 0.0014004199999817502, | |
| "count": 8, | |
| "is_parallel": true, | |
| "self": 0.0014004199999817502 | |
| } | |
| } | |
| }, | |
| "UnityEnvironment.step": { | |
| "total": 0.06185051000011299, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0006236980002540804, | |
| "children": { | |
| "UnityEnvironment._generate_step_input": { | |
| "total": 0.0005590010000560142, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0005590010000560142 | |
| }, | |
| "communicator.exchange": { | |
| "total": 0.058660017999955016, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.058660017999955016 | |
| }, | |
| "steps_from_proto": { | |
| "total": 0.0020077929998478794, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0004621249991032528, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 0.0015456680007446266, | |
| "count": 8, | |
| "is_parallel": true, | |
| "self": 0.0015456680007446266 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "UnityEnvironment.step": { | |
| "total": 6574.276149818184, | |
| "count": 194775, | |
| "is_parallel": true, | |
| "self": 126.22423868476926, | |
| "children": { | |
| "UnityEnvironment._generate_step_input": { | |
| "total": 87.66725014712665, | |
| "count": 194775, | |
| "is_parallel": true, | |
| "self": 87.66725014712665 | |
| }, | |
| "communicator.exchange": { | |
| "total": 5939.873369345933, | |
| "count": 194775, | |
| "is_parallel": true, | |
| "self": 5939.873369345933 | |
| }, | |
| "steps_from_proto": { | |
| "total": 420.5112916403559, | |
| "count": 194775, | |
| "is_parallel": true, | |
| "self": 84.27840419445602, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 336.2328874458999, | |
| "count": 1558200, | |
| "is_parallel": true, | |
| "self": 336.2328874458999 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "trainer_advance": { | |
| "total": 2281.489277852118, | |
| "count": 194776, | |
| "self": 11.9832699893077, | |
| "children": { | |
| "process_trajectory": { | |
| "total": 417.47297203381004, | |
| "count": 194776, | |
| "self": 416.8497563738106, | |
| "children": { | |
| "RLTrainer._checkpoint": { | |
| "total": 0.6232156599994596, | |
| "count": 6, | |
| "self": 0.6232156599994596 | |
| } | |
| } | |
| }, | |
| "_update_policy": { | |
| "total": 1852.0330358290003, | |
| "count": 1394, | |
| "self": 989.4156551500228, | |
| "children": { | |
| "TorchPPOOptimizer.update": { | |
| "total": 862.6173806789775, | |
| "count": 68373, | |
| "self": 862.6173806789775 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "trainer_threads": { | |
| "total": 1.22199890029151e-06, | |
| "count": 1, | |
| "self": 1.22199890029151e-06 | |
| }, | |
| "TrainerController._save_models": { | |
| "total": 0.08054240199999185, | |
| "count": 1, | |
| "self": 0.0010554749987932155, | |
| "children": { | |
| "RLTrainer._checkpoint": { | |
| "total": 0.07948692700119864, | |
| "count": 1, | |
| "self": 0.07948692700119864 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } |