Reinforcement Learning
ml-agents
TensorBoard
ONNX
Pyramids
deep-reinforcement-learning
ML-Agents-Pyramids
Instructions to use vicsonsam/ppo-PyramidsRND with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- ml-agents
How to use vicsonsam/ppo-PyramidsRND with ml-agents:
mlagents-load-from-hf --repo-id="vicsonsam/ppo-PyramidsRND" --local-dir="./download: string[]s"
- Notebooks
- Google Colab
- Kaggle
| { | |
| "name": "root", | |
| "gauges": { | |
| "Pyramids.Policy.Entropy.mean": { | |
| "value": 0.6362375617027283, | |
| "min": 0.6313311457633972, | |
| "max": 1.4627357721328735, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Entropy.sum": { | |
| "value": 18883.53125, | |
| "min": 18883.53125, | |
| "max": 44373.55078125, | |
| "count": 33 | |
| }, | |
| "Pyramids.Step.mean": { | |
| "value": 989984.0, | |
| "min": 29924.0, | |
| "max": 989984.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Step.sum": { | |
| "value": 989984.0, | |
| "min": 29924.0, | |
| "max": 989984.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicValueEstimate.mean": { | |
| "value": 0.2534390091896057, | |
| "min": -0.12295710295438766, | |
| "max": 0.2885737419128418, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicValueEstimate.sum": { | |
| "value": 65.64070129394531, | |
| "min": -29.50970458984375, | |
| "max": 75.6063232421875, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndValueEstimate.mean": { | |
| "value": -0.01804141141474247, | |
| "min": -0.01804141141474247, | |
| "max": 0.16768185794353485, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndValueEstimate.sum": { | |
| "value": -4.672725677490234, | |
| "min": -4.672725677490234, | |
| "max": 40.41132736206055, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.PolicyLoss.mean": { | |
| "value": 0.07194950870384662, | |
| "min": 0.06615968040822932, | |
| "max": 0.07365789367364872, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.PolicyLoss.sum": { | |
| "value": 1.0072931218538526, | |
| "min": 0.515605255715541, | |
| "max": 1.0478990673921849, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.ValueLoss.mean": { | |
| "value": 0.012404981810273144, | |
| "min": 6.899401456621187e-05, | |
| "max": 0.012404981810273144, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.ValueLoss.sum": { | |
| "value": 0.173669745343824, | |
| "min": 0.0009659162039269662, | |
| "max": 0.173669745343824, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.LearningRate.mean": { | |
| "value": 7.47354750885e-06, | |
| "min": 7.47354750885e-06, | |
| "max": 0.00029523411587434285, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.LearningRate.sum": { | |
| "value": 0.0001046296651239, | |
| "min": 0.0001046296651239, | |
| "max": 0.0035082668305777996, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Epsilon.mean": { | |
| "value": 0.10249114999999999, | |
| "min": 0.10249114999999999, | |
| "max": 0.1984113714285714, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Epsilon.sum": { | |
| "value": 1.4348760999999999, | |
| "min": 1.3888795999999999, | |
| "max": 2.5694222, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Beta.mean": { | |
| "value": 0.00025886588500000004, | |
| "min": 0.00025886588500000004, | |
| "max": 0.009841296005714286, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.Beta.sum": { | |
| "value": 0.0036241223900000006, | |
| "min": 0.0036241223900000006, | |
| "max": 0.11696527778, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.RNDLoss.mean": { | |
| "value": 0.01060565747320652, | |
| "min": 0.01040094904601574, | |
| "max": 0.35427170991897583, | |
| "count": 33 | |
| }, | |
| "Pyramids.Losses.RNDLoss.sum": { | |
| "value": 0.14847920835018158, | |
| "min": 0.14561328291893005, | |
| "max": 2.4799020290374756, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.EpisodeLength.mean": { | |
| "value": 528.4137931034483, | |
| "min": 528.4137931034483, | |
| "max": 999.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.EpisodeLength.sum": { | |
| "value": 30648.0, | |
| "min": 16723.0, | |
| "max": 32732.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.CumulativeReward.mean": { | |
| "value": 1.1611379030449638, | |
| "min": -0.9997852378421359, | |
| "max": 1.1611379030449638, | |
| "count": 33 | |
| }, | |
| "Pyramids.Environment.CumulativeReward.sum": { | |
| "value": 67.3459983766079, | |
| "min": -30.76280176639557, | |
| "max": 67.3459983766079, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicReward.mean": { | |
| "value": 1.1611379030449638, | |
| "min": -0.9997852378421359, | |
| "max": 1.1611379030449638, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.ExtrinsicReward.sum": { | |
| "value": 67.3459983766079, | |
| "min": -30.76280176639557, | |
| "max": 67.3459983766079, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndReward.mean": { | |
| "value": 0.05816971184184036, | |
| "min": 0.05816971184184036, | |
| "max": 6.880389970891616, | |
| "count": 33 | |
| }, | |
| "Pyramids.Policy.RndReward.sum": { | |
| "value": 3.373843286826741, | |
| "min": 3.1480140782514354, | |
| "max": 116.96662950515747, | |
| "count": 33 | |
| }, | |
| "Pyramids.IsTraining.mean": { | |
| "value": 1.0, | |
| "min": 1.0, | |
| "max": 1.0, | |
| "count": 33 | |
| }, | |
| "Pyramids.IsTraining.sum": { | |
| "value": 1.0, | |
| "min": 1.0, | |
| "max": 1.0, | |
| "count": 33 | |
| } | |
| }, | |
| "metadata": { | |
| "timer_format_version": "0.1.0", | |
| "start_time_seconds": "1784336983", | |
| "python_version": "3.10.12 (main, Jul 5 2023, 18:54:27) [GCC 11.2.0]", | |
| "command_line_arguments": "/usr/local/bin/mlagents-learn ./config/ppo/PyramidsRND.yaml --env=./training-envs-executables/linux/Pyramids/Pyramids --run-id=Pyramids Training --no-graphics", | |
| "mlagents_version": "1.2.0.dev0", | |
| "mlagents_envs_version": "1.2.0.dev0", | |
| "communication_protocol_version": "1.5.0", | |
| "pytorch_version": "2.8.0+cu128", | |
| "numpy_version": "1.23.5", | |
| "end_time_seconds": "1784338562" | |
| }, | |
| "total": 1579.2364225259998, | |
| "count": 1, | |
| "self": 0.32156133799981035, | |
| "children": { | |
| "run_training.setup": { | |
| "total": 0.02468523799984723, | |
| "count": 1, | |
| "self": 0.02468523799984723 | |
| }, | |
| "TrainerController.start_learning": { | |
| "total": 1578.8901759500002, | |
| "count": 1, | |
| "self": 1.242672765035195, | |
| "children": { | |
| "TrainerController._reset_env": { | |
| "total": 2.367709945999877, | |
| "count": 1, | |
| "self": 2.367709945999877 | |
| }, | |
| "TrainerController.advance": { | |
| "total": 1575.2062154749653, | |
| "count": 63388, | |
| "self": 1.1870749749882634, | |
| "children": { | |
| "env_step": { | |
| "total": 984.4209435319897, | |
| "count": 63388, | |
| "self": 844.9678011699693, | |
| "children": { | |
| "SubprocessEnvManager._take_step": { | |
| "total": 138.7116066899814, | |
| "count": 63388, | |
| "self": 4.184227928986047, | |
| "children": { | |
| "TorchPolicy.evaluate": { | |
| "total": 134.52737876099536, | |
| "count": 62567, | |
| "self": 134.52737876099536 | |
| } | |
| } | |
| }, | |
| "workers": { | |
| "total": 0.7415356720389354, | |
| "count": 63388, | |
| "self": 0.0, | |
| "children": { | |
| "worker_root": { | |
| "total": 1576.6757748989498, | |
| "count": 63388, | |
| "is_parallel": true, | |
| "self": 819.2188703679444, | |
| "children": { | |
| "run_training.setup": { | |
| "total": 0.0, | |
| "count": 0, | |
| "is_parallel": true, | |
| "self": 0.0, | |
| "children": { | |
| "steps_from_proto": { | |
| "total": 0.0018781759999910719, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0006271819997891726, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 0.0012509940002018993, | |
| "count": 8, | |
| "is_parallel": true, | |
| "self": 0.0012509940002018993 | |
| } | |
| } | |
| }, | |
| "UnityEnvironment.step": { | |
| "total": 0.03484569499983081, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0003420819996335922, | |
| "children": { | |
| "UnityEnvironment._generate_step_input": { | |
| "total": 0.0003207590000329219, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0003207590000329219 | |
| }, | |
| "communicator.exchange": { | |
| "total": 0.03316115400002673, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.03316115400002673 | |
| }, | |
| "steps_from_proto": { | |
| "total": 0.001021700000137571, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.00023027499992167577, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 0.0007914250002158951, | |
| "count": 8, | |
| "is_parallel": true, | |
| "self": 0.0007914250002158951 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "UnityEnvironment.step": { | |
| "total": 757.4569045310054, | |
| "count": 63387, | |
| "is_parallel": true, | |
| "self": 20.013743361062552, | |
| "children": { | |
| "UnityEnvironment._generate_step_input": { | |
| "total": 13.483533401980822, | |
| "count": 63387, | |
| "is_parallel": true, | |
| "self": 13.483533401980822 | |
| }, | |
| "communicator.exchange": { | |
| "total": 666.6476047939973, | |
| "count": 63387, | |
| "is_parallel": true, | |
| "self": 666.6476047939973 | |
| }, | |
| "steps_from_proto": { | |
| "total": 57.31202297396476, | |
| "count": 63387, | |
| "is_parallel": true, | |
| "self": 12.217451158930317, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 45.09457181503444, | |
| "count": 507096, | |
| "is_parallel": true, | |
| "self": 45.09457181503444 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "trainer_advance": { | |
| "total": 589.5981969679874, | |
| "count": 63388, | |
| "self": 2.382095324048578, | |
| "children": { | |
| "process_trajectory": { | |
| "total": 100.31896107794546, | |
| "count": 63388, | |
| "self": 100.14229111394525, | |
| "children": { | |
| "RLTrainer._checkpoint": { | |
| "total": 0.17666996400021162, | |
| "count": 2, | |
| "self": 0.17666996400021162 | |
| } | |
| } | |
| }, | |
| "_update_policy": { | |
| "total": 486.89714056599337, | |
| "count": 451, | |
| "self": 256.99970595397326, | |
| "children": { | |
| "TorchPPOOptimizer.update": { | |
| "total": 229.8974346120201, | |
| "count": 22764, | |
| "self": 229.8974346120201 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "trainer_threads": { | |
| "total": 9.919999683916103e-07, | |
| "count": 1, | |
| "self": 9.919999683916103e-07 | |
| }, | |
| "TrainerController._save_models": { | |
| "total": 0.07357677199979662, | |
| "count": 1, | |
| "self": 0.0009788159995878232, | |
| "children": { | |
| "RLTrainer._checkpoint": { | |
| "total": 0.0725979560002088, | |
| "count": 1, | |
| "self": 0.0725979560002088 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } |