Reinforcement Learning
ml-agents
TensorBoard
ONNX
Pyramids
deep-reinforcement-learning
ML-Agents-Pyramids
Instructions to use quiago/ppo-PyramidsTraining with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- ml-agents
How to use quiago/ppo-PyramidsTraining with ml-agents:
mlagents-load-from-hf --repo-id="quiago/ppo-PyramidsTraining" --local-dir="./download: string[]s"
- Notebooks
- Google Colab
- Kaggle
| { | |
| "name": "root", | |
| "gauges": { | |
| "Pyramids.Policy.Entropy.mean": { | |
| "value": 0.16093112528324127, | |
| "min": 0.14973129332065582, | |
| "max": 1.4687169790267944, | |
| "count": 73 | |
| }, | |
| "Pyramids.Policy.Entropy.sum": { | |
| "value": 4827.93359375, | |
| "min": 4465.5859375, | |
| "max": 44555.0, | |
| "count": 73 | |
| }, | |
| "Pyramids.Step.mean": { | |
| "value": 2189999.0, | |
| "min": 29952.0, | |
| "max": 2189999.0, | |
| "count": 73 | |
| }, | |
| "Pyramids.Step.sum": { | |
| "value": 2189999.0, | |
| "min": 29952.0, | |
| "max": 2189999.0, | |
| "count": 73 | |
| }, | |
| "Pyramids.Policy.ExtrinsicValueEstimate.mean": { | |
| "value": 0.7698275446891785, | |
| "min": -0.11906354874372482, | |
| "max": 0.8018506169319153, | |
| "count": 73 | |
| }, | |
| "Pyramids.Policy.ExtrinsicValueEstimate.sum": { | |
| "value": 227.86895751953125, | |
| "min": -28.575252532958984, | |
| "max": 234.140380859375, | |
| "count": 73 | |
| }, | |
| "Pyramids.Policy.RndValueEstimate.mean": { | |
| "value": 0.012321759946644306, | |
| "min": -0.008147596381604671, | |
| "max": 0.2541937530040741, | |
| "count": 73 | |
| }, | |
| "Pyramids.Policy.RndValueEstimate.sum": { | |
| "value": 3.6472408771514893, | |
| "min": -2.1265225410461426, | |
| "max": 61.006500244140625, | |
| "count": 73 | |
| }, | |
| "Pyramids.Losses.PolicyLoss.mean": { | |
| "value": 0.07202943201090896, | |
| "min": 0.06555995500853493, | |
| "max": 0.07514585461059888, | |
| "count": 73 | |
| }, | |
| "Pyramids.Losses.PolicyLoss.sum": { | |
| "value": 1.0084120481527254, | |
| "min": 0.4884751299690541, | |
| "max": 1.0711755824935003, | |
| "count": 73 | |
| }, | |
| "Pyramids.Losses.ValueLoss.mean": { | |
| "value": 0.01305162610792433, | |
| "min": 0.00018490804629228415, | |
| "max": 0.016653739772866376, | |
| "count": 73 | |
| }, | |
| "Pyramids.Losses.ValueLoss.sum": { | |
| "value": 0.1827227655109406, | |
| "min": 0.0022188965555074098, | |
| "max": 0.24980609659299563, | |
| "count": 73 | |
| }, | |
| "Pyramids.Policy.LearningRate.mean": { | |
| "value": 8.245804394258094e-05, | |
| "min": 8.245804394258094e-05, | |
| "max": 0.00029838354339596195, | |
| "count": 73 | |
| }, | |
| "Pyramids.Policy.LearningRate.sum": { | |
| "value": 0.0011544126151961332, | |
| "min": 0.0011544126151961332, | |
| "max": 0.004027430757523133, | |
| "count": 73 | |
| }, | |
| "Pyramids.Policy.Epsilon.mean": { | |
| "value": 0.1274859904761905, | |
| "min": 0.1274859904761905, | |
| "max": 0.19946118095238097, | |
| "count": 73 | |
| }, | |
| "Pyramids.Policy.Epsilon.sum": { | |
| "value": 1.7848038666666668, | |
| "min": 1.3897045333333333, | |
| "max": 2.842476866666667, | |
| "count": 73 | |
| }, | |
| "Pyramids.Policy.Beta.mean": { | |
| "value": 0.0027558504485714294, | |
| "min": 0.0027558504485714294, | |
| "max": 0.009946171977142856, | |
| "count": 73 | |
| }, | |
| "Pyramids.Policy.Beta.sum": { | |
| "value": 0.03858190628000001, | |
| "min": 0.03858190628000001, | |
| "max": 0.13426343898, | |
| "count": 73 | |
| }, | |
| "Pyramids.Losses.RNDLoss.mean": { | |
| "value": 0.008542558178305626, | |
| "min": 0.00810875091701746, | |
| "max": 0.34159430861473083, | |
| "count": 73 | |
| }, | |
| "Pyramids.Losses.RNDLoss.sum": { | |
| "value": 0.11959581822156906, | |
| "min": 0.11352250725030899, | |
| "max": 2.391160249710083, | |
| "count": 73 | |
| }, | |
| "Pyramids.Environment.EpisodeLength.mean": { | |
| "value": 238.94827586206895, | |
| "min": 237.88429752066116, | |
| "max": 999.0, | |
| "count": 73 | |
| }, | |
| "Pyramids.Environment.EpisodeLength.sum": { | |
| "value": 27718.0, | |
| "min": 15984.0, | |
| "max": 34345.0, | |
| "count": 73 | |
| }, | |
| "Pyramids.Environment.CumulativeReward.mean": { | |
| "value": 1.7265517061640476, | |
| "min": -1.0000000521540642, | |
| "max": 1.7469649847596884, | |
| "count": 73 | |
| }, | |
| "Pyramids.Environment.CumulativeReward.sum": { | |
| "value": 200.27999791502953, | |
| "min": -32.000001668930054, | |
| "max": 215.98819863796234, | |
| "count": 73 | |
| }, | |
| "Pyramids.Policy.ExtrinsicReward.mean": { | |
| "value": 1.7265517061640476, | |
| "min": -1.0000000521540642, | |
| "max": 1.7469649847596884, | |
| "count": 73 | |
| }, | |
| "Pyramids.Policy.ExtrinsicReward.sum": { | |
| "value": 200.27999791502953, | |
| "min": -32.000001668930054, | |
| "max": 215.98819863796234, | |
| "count": 73 | |
| }, | |
| "Pyramids.Policy.RndReward.mean": { | |
| "value": 0.02134822721012009, | |
| "min": 0.021039137857709042, | |
| "max": 6.199004173278809, | |
| "count": 73 | |
| }, | |
| "Pyramids.Policy.RndReward.sum": { | |
| "value": 2.4763943563739303, | |
| "min": 2.4763943563739303, | |
| "max": 99.18406677246094, | |
| "count": 73 | |
| }, | |
| "Pyramids.IsTraining.mean": { | |
| "value": 1.0, | |
| "min": 1.0, | |
| "max": 1.0, | |
| "count": 73 | |
| }, | |
| "Pyramids.IsTraining.sum": { | |
| "value": 1.0, | |
| "min": 1.0, | |
| "max": 1.0, | |
| "count": 73 | |
| } | |
| }, | |
| "metadata": { | |
| "timer_format_version": "0.1.0", | |
| "start_time_seconds": "1777362201", | |
| "python_version": "3.10.12 (main, Jul 5 2023, 18:54:27) [GCC 11.2.0]", | |
| "command_line_arguments": "/usr/local/bin/mlagents-learn ./config/ppo/PyramidsRND.yaml --env=./training-envs-executables/linux/Pyramids/Pyramids --run-id=Pyramids Training --no-graphics", | |
| "mlagents_version": "1.2.0.dev0", | |
| "mlagents_envs_version": "1.2.0.dev0", | |
| "communication_protocol_version": "1.5.0", | |
| "pytorch_version": "2.8.0+cu128", | |
| "numpy_version": "1.23.5", | |
| "end_time_seconds": "1777367616" | |
| }, | |
| "total": 5414.665922715, | |
| "count": 1, | |
| "self": 0.3932694530003573, | |
| "children": { | |
| "run_training.setup": { | |
| "total": 0.02395800399995096, | |
| "count": 1, | |
| "self": 0.02395800399995096 | |
| }, | |
| "TrainerController.start_learning": { | |
| "total": 5414.248695258, | |
| "count": 1, | |
| "self": 3.3579498651433823, | |
| "children": { | |
| "TrainerController._reset_env": { | |
| "total": 2.982527981000203, | |
| "count": 1, | |
| "self": 2.982527981000203 | |
| }, | |
| "TrainerController.advance": { | |
| "total": 5407.718882891856, | |
| "count": 143167, | |
| "self": 3.4943771617954553, | |
| "children": { | |
| "env_step": { | |
| "total": 3898.51154172517, | |
| "count": 143167, | |
| "self": 3539.6839358451025, | |
| "children": { | |
| "SubprocessEnvManager._take_step": { | |
| "total": 356.768004589165, | |
| "count": 143167, | |
| "self": 10.675336042209892, | |
| "children": { | |
| "TorchPolicy.evaluate": { | |
| "total": 346.0926685469551, | |
| "count": 138456, | |
| "self": 346.0926685469551 | |
| } | |
| } | |
| }, | |
| "workers": { | |
| "total": 2.0596012909022647, | |
| "count": 143167, | |
| "self": 0.0, | |
| "children": { | |
| "worker_root": { | |
| "total": 5397.7861618759935, | |
| "count": 143167, | |
| "is_parallel": true, | |
| "self": 2136.8680878639325, | |
| "children": { | |
| "run_training.setup": { | |
| "total": 0.0, | |
| "count": 0, | |
| "is_parallel": true, | |
| "self": 0.0, | |
| "children": { | |
| "steps_from_proto": { | |
| "total": 0.0024737400001413334, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0007250270002714387, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 0.0017487129998698947, | |
| "count": 8, | |
| "is_parallel": true, | |
| "self": 0.0017487129998698947 | |
| } | |
| } | |
| }, | |
| "UnityEnvironment.step": { | |
| "total": 0.07875738300003832, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0006336239998745441, | |
| "children": { | |
| "UnityEnvironment._generate_step_input": { | |
| "total": 0.0016093780000119295, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0016093780000119295 | |
| }, | |
| "communicator.exchange": { | |
| "total": 0.0745510760000343, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0745510760000343 | |
| }, | |
| "steps_from_proto": { | |
| "total": 0.0019633050001175434, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0005785980001746793, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 0.001384706999942864, | |
| "count": 8, | |
| "is_parallel": true, | |
| "self": 0.001384706999942864 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "UnityEnvironment.step": { | |
| "total": 3260.918074012061, | |
| "count": 143166, | |
| "is_parallel": true, | |
| "self": 77.39514849019542, | |
| "children": { | |
| "UnityEnvironment._generate_step_input": { | |
| "total": 53.32608670487957, | |
| "count": 143166, | |
| "is_parallel": true, | |
| "self": 53.32608670487957 | |
| }, | |
| "communicator.exchange": { | |
| "total": 2877.7350703080283, | |
| "count": 143166, | |
| "is_parallel": true, | |
| "self": 2877.7350703080283 | |
| }, | |
| "steps_from_proto": { | |
| "total": 252.4617685089579, | |
| "count": 143166, | |
| "is_parallel": true, | |
| "self": 53.73386885277728, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 198.72789965618063, | |
| "count": 1145328, | |
| "is_parallel": true, | |
| "self": 198.72789965618063 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "trainer_advance": { | |
| "total": 1505.7129640048902, | |
| "count": 143167, | |
| "self": 6.665443870746458, | |
| "children": { | |
| "process_trajectory": { | |
| "total": 288.9727525321432, | |
| "count": 143167, | |
| "self": 288.6280877711433, | |
| "children": { | |
| "RLTrainer._checkpoint": { | |
| "total": 0.3446647609998763, | |
| "count": 4, | |
| "self": 0.3446647609998763 | |
| } | |
| } | |
| }, | |
| "_update_policy": { | |
| "total": 1210.0747676020005, | |
| "count": 1023, | |
| "self": 669.363920580183, | |
| "children": { | |
| "TorchPPOOptimizer.update": { | |
| "total": 540.7108470218175, | |
| "count": 50478, | |
| "self": 540.7108470218175 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "trainer_threads": { | |
| "total": 1.4960005501052365e-06, | |
| "count": 1, | |
| "self": 1.4960005501052365e-06 | |
| }, | |
| "TrainerController._save_models": { | |
| "total": 0.18933302400000684, | |
| "count": 1, | |
| "self": 0.0013314219995663734, | |
| "children": { | |
| "RLTrainer._checkpoint": { | |
| "total": 0.18800160200044047, | |
| "count": 1, | |
| "self": 0.18800160200044047 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } |