Reinforcement Learning
ml-agents
TensorBoard
ONNX
Pyramids
deep-reinforcement-learning
ML-Agents-Pyramids
Instructions to use Akshaykumar4321/ppo-Pyramids with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- ml-agents
How to use Akshaykumar4321/ppo-Pyramids with ml-agents:
mlagents-load-from-hf --repo-id="Akshaykumar4321/ppo-Pyramids" --local-dir="./download: string[]s"
- Notebooks
- Google Colab
- Kaggle
| { | |
| "name": "root", | |
| "gauges": { | |
| "Pyramids.Policy.Entropy.mean": { | |
| "value": 0.48063433170318604, | |
| "min": 0.4692344069480896, | |
| "max": 1.470839023590088, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Entropy.sum": { | |
| "value": 14388.26953125, | |
| "min": 14049.02734375, | |
| "max": 44619.37109375, | |
| "count": 33 | |
| }, | |
| "Pyramids.Step.mean": { | |
| "value": 989977.0, | |
| "min": 29952.0, | |
| "max": 989977.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Step.sum": { | |
| "value": 989977.0, | |
| "min": 29952.0, | |
| "max": 989977.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicValueEstimate.mean": { | |
| "value": 0.44532668590545654, | |
| "min": -0.09884803742170334, | |
| "max": 0.5579422116279602, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicValueEstimate.sum": { | |
| "value": 122.01951599121094, | |
| "min": -23.822376251220703, | |
| "max": 156.22381591796875, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndValueEstimate.mean": { | |
| "value": -0.02150287851691246, | |
| "min": -0.02150287851691246, | |
| "max": 0.21008716523647308, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndValueEstimate.sum": { | |
| "value": -5.891788482666016, | |
| "min": -5.891788482666016, | |
| "max": 50.420921325683594, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.PolicyLoss.mean": { | |
| "value": 0.06535600314895247, | |
| "min": 0.06429562698604302, | |
| "max": 0.07519610611285459, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.PolicyLoss.sum": { | |
| "value": 0.9149840440853345, | |
| "min": 0.4851201793771697, | |
| "max": 1.0769569803575425, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.ValueLoss.mean": { | |
| "value": 0.014496754247548304, | |
| "min": 0.0004073721338207389, | |
| "max": 0.014873691412503831, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.ValueLoss.sum": { | |
| "value": 0.20295455946567625, | |
| "min": 0.005295837739669606, | |
| "max": 0.22310537118755747, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.LearningRate.mean": { | |
| "value": 7.700668861714284e-06, | |
| "min": 7.700668861714284e-06, | |
| "max": 0.00029515063018788575, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.LearningRate.sum": { | |
| "value": 0.00010780936406399997, | |
| "min": 0.00010780936406399997, | |
| "max": 0.003371705276098299, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Epsilon.mean": { | |
| "value": 0.10256685714285715, | |
| "min": 0.10256685714285715, | |
| "max": 0.19838354285714285, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Epsilon.sum": { | |
| "value": 1.435936, | |
| "min": 1.3691136000000002, | |
| "max": 2.4424554000000005, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Beta.mean": { | |
| "value": 0.00026642902857142864, | |
| "min": 0.00026642902857142864, | |
| "max": 0.00983851593142857, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Beta.sum": { | |
| "value": 0.003730006400000001, | |
| "min": 0.003730006400000001, | |
| "max": 0.11240777982999998, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.RNDLoss.mean": { | |
| "value": 0.010170450434088707, | |
| "min": 0.010170450434088707, | |
| "max": 0.3285984396934509, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.RNDLoss.sum": { | |
| "value": 0.1423863023519516, | |
| "min": 0.1423863023519516, | |
| "max": 2.3001890182495117, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.EpisodeLength.mean": { | |
| "value": 438.69444444444446, | |
| "min": 347.86746987951807, | |
| "max": 999.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.EpisodeLength.sum": { | |
| "value": 31586.0, | |
| "min": 15984.0, | |
| "max": 33630.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.CumulativeReward.mean": { | |
| "value": 1.4223833130672574, | |
| "min": -1.0000000521540642, | |
| "max": 1.603932511375611, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.CumulativeReward.sum": { | |
| "value": 102.41159854084253, | |
| "min": -32.000001668930054, | |
| "max": 133.12639844417572, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicReward.mean": { | |
| "value": 1.4223833130672574, | |
| "min": -1.0000000521540642, | |
| "max": 1.603932511375611, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicReward.sum": { | |
| "value": 102.41159854084253, | |
| "min": -32.000001668930054, | |
| "max": 133.12639844417572, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndReward.mean": { | |
| "value": 0.04596105357551197, | |
| "min": 0.0393430610749792, | |
| "max": 6.373056381009519, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndReward.sum": { | |
| "value": 3.309195857436862, | |
| "min": 3.2654740692232735, | |
| "max": 101.9689020961523, | |
| "count": 33 | |
| }, | |
| "Pyramids.IsTraining.mean": { | |
| "value": 1.0, | |
| "min": 1.0, | |
| "max": 1.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.IsTraining.sum": { | |
| "value": 1.0, | |
| "min": 1.0, | |
| "max": 1.0, | |
| "count": 33 | |
| } | |
| }, | |
| "metadata": { | |
| "timer_format_version": "0.1.0", | |
| "start_time_seconds": "1788610183", | |
| "python_version": "3.10.12 (main, Jul 5 2023, 18:54:27) [GCC 11.2.0]", | |
| "command_line_arguments": "/usr/local/envs/mlagents_env/bin/mlagents-learn ./ml-agents/config/ppo/PyramidsRND.yaml --env=./training-envs-executables/linux/Pyramids/Pyramids --run-id=Pyramids_Training --no-graphics --force", | |
| "mlagents_version": "1.2.0.dev0", | |
| "mlagents_envs_version": "1.2.0.dev0", | |
| "communication_protocol_version": "1.5.0", | |
| "pytorch_version": "2.8.0+cu128", | |
| "numpy_version": "1.23.5", | |
| "end_time_seconds": "1788612640" | |
| }, | |
| "total": 2457.3661236489997, | |
| "count": 1, | |
| "self": 0.4801017789995967, | |
| "children": { | |
| "run_training.setup": { | |
| "total": 0.019352670999978727, | |
| "count": 1, | |
| "self": 0.019352670999978727 | |
| }, | |
| "TrainerController.start_learning": { | |
| "total": 2456.866669199, | |
| "count": 1, | |
| "self": 1.6697714270039796, | |
| "children": { | |
| "TrainerController._reset_env": { | |
| "total": 3.1423468160000994, | |
| "count": 1, | |
| "self": 3.1423468160000994 | |
| }, | |
| "TrainerController.advance": { | |
| "total": 2451.972584205996, | |
| "count": 63645, | |
| "self": 1.644876678069977, | |
| "children": { | |
| "env_step": { | |
| "total": 1790.0176028909702, | |
| "count": 63645, | |
| "self": 1619.0932308129154, | |
| "children": { | |
| "SubprocessEnvManager._take_step": { | |
| "total": 169.93236793501342, | |
| "count": 63645, | |
| "self": 5.099558166002453, | |
| "children": { | |
| "TorchPolicy.evaluate": { | |
| "total": 164.83280976901096, | |
| "count": 62567, | |
| "self": 164.83280976901096 | |
| } | |
| } | |
| }, | |
| "workers": { | |
| "total": 0.992004143041413, | |
| "count": 63645, | |
| "self": 0.0, | |
| "children": { | |
| "worker_root": { | |
| "total": 2450.9944483749773, | |
| "count": 63645, | |
| "is_parallel": true, | |
| "self": 963.7221945030533, | |
| "children": { | |
| "run_training.setup": { | |
| "total": 0.0, | |
| "count": 0, | |
| "is_parallel": true, | |
| "self": 0.0, | |
| "children": { | |
| "steps_from_proto": { | |
| "total": 0.004482126000084463, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0031781590000719007, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 0.0013039670000125625, | |
| "count": 8, | |
| "is_parallel": true, | |
| "self": 0.0013039670000125625 | |
| } | |
| } | |
| }, | |
| "UnityEnvironment.step": { | |
| "total": 0.05009932899997693, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0005394040001647227, | |
| "children": { | |
| "UnityEnvironment._generate_step_input": { | |
| "total": 0.0004750479999984236, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0004750479999984236 | |
| }, | |
| "communicator.exchange": { | |
| "total": 0.0472565359998498, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0472565359998498 | |
| }, | |
| "steps_from_proto": { | |
| "total": 0.0018283409999639844, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.00037147699981687765, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 0.0014568640001471067, | |
| "count": 8, | |
| "is_parallel": true, | |
| "self": 0.0014568640001471067 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "UnityEnvironment.step": { | |
| "total": 1487.272253871924, | |
| "count": 63644, | |
| "is_parallel": true, | |
| "self": 36.17591552593467, | |
| "children": { | |
| "UnityEnvironment._generate_step_input": { | |
| "total": 24.96874212199691, | |
| "count": 63644, | |
| "is_parallel": true, | |
| "self": 24.96874212199691 | |
| }, | |
| "communicator.exchange": { | |
| "total": 1307.4988795109437, | |
| "count": 63644, | |
| "is_parallel": true, | |
| "self": 1307.4988795109437 | |
| }, | |
| "steps_from_proto": { | |
| "total": 118.62871671304879, | |
| "count": 63644, | |
| "is_parallel": true, | |
| "self": 25.104068238325908, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 93.52464847472288, | |
| "count": 509152, | |
| "is_parallel": true, | |
| "self": 93.52464847472288 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "trainer_advance": { | |
| "total": 660.3101046369559, | |
| "count": 63645, | |
| "self": 3.148322349987666, | |
| "children": { | |
| "process_trajectory": { | |
| "total": 118.57478626796637, | |
| "count": 63645, | |
| "self": 118.32724236396643, | |
| "children": { | |
| "RLTrainer._checkpoint": { | |
| "total": 0.24754390399994008, | |
| "count": 2, | |
| "self": 0.24754390399994008 | |
| } | |
| } | |
| }, | |
| "_update_policy": { | |
| "total": 538.5869960190018, | |
| "count": 444, | |
| "self": 289.10093802799565, | |
| "children": { | |
| "TorchPPOOptimizer.update": { | |
| "total": 249.4860579910062, | |
| "count": 22791, | |
| "self": 249.4860579910062 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "trainer_threads": { | |
| "total": 8.889996934158262e-07, | |
| "count": 1, | |
| "self": 8.889996934158262e-07 | |
| }, | |
| "TrainerController._save_models": { | |
| "total": 0.0819658610002989, | |
| "count": 1, | |
| "self": 0.000925504000406363, | |
| "children": { | |
| "RLTrainer._checkpoint": { | |
| "total": 0.08104035699989254, | |
| "count": 1, | |
| "self": 0.08104035699989254 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } |