Reinforcement Learning
ml-agents
TensorBoard
ONNX
Pyramids
deep-reinforcement-learning
ML-Agents-Pyramids
Instructions to use fanjiangpost/pyramids with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- ml-agents
How to use fanjiangpost/pyramids with ml-agents:
mlagents-load-from-hf --repo-id="fanjiangpost/pyramids" --local-dir="./download: string[]s"
- Notebooks
- Google Colab
- Kaggle
| { | |
| "name": "root", | |
| "gauges": { | |
| "Pyramids.Policy.Entropy.mean": { | |
| "value": 0.9259867668151855, | |
| "min": 0.8769648671150208, | |
| "max": 0.9259867668151855, | |
| "count": 3 | |
| }, | |
| "Pyramids.Policy.Entropy.sum": { | |
| "value": 27735.15625, | |
| "min": 17623.486328125, | |
| "max": 27735.15625, | |
| "count": 3 | |
| }, | |
| "Pyramids.Step.mean": { | |
| "value": 179921.0, | |
| "min": 119888.0, | |
| "max": 179921.0, | |
| "count": 3 | |
| }, | |
| "Pyramids.Step.sum": { | |
| "value": 179921.0, | |
| "min": 119888.0, | |
| "max": 179921.0, | |
| "count": 3 | |
| }, | |
| "Pyramids.Policy.ExtrinsicValueEstimate.mean": { | |
| "value": -0.09521700441837311, | |
| "min": -0.10461809486150742, | |
| "max": -0.09216971695423126, | |
| "count": 3 | |
| }, | |
| "Pyramids.Policy.ExtrinsicValueEstimate.sum": { | |
| "value": -22.852081298828125, | |
| "min": -22.852081298828125, | |
| "max": -16.529659271240234, | |
| "count": 3 | |
| }, | |
| "Pyramids.Policy.RndValueEstimate.mean": { | |
| "value": 0.06673477590084076, | |
| "min": 0.06673477590084076, | |
| "max": 0.10928431898355484, | |
| "count": 3 | |
| }, | |
| "Pyramids.Policy.RndValueEstimate.sum": { | |
| "value": 16.016345977783203, | |
| "min": 16.016345977783203, | |
| "max": 23.216726303100586, | |
| "count": 3 | |
| }, | |
| "Pyramids.Losses.PolicyLoss.mean": { | |
| "value": 0.06650441241170575, | |
| "min": 0.06650441241170575, | |
| "max": 0.07030591471725726, | |
| "count": 3 | |
| }, | |
| "Pyramids.Losses.PolicyLoss.sum": { | |
| "value": 0.798052948940469, | |
| "min": 0.2801994997298816, | |
| "max": 0.798052948940469, | |
| "count": 3 | |
| }, | |
| "Pyramids.Losses.ValueLoss.mean": { | |
| "value": 0.00017486471076749626, | |
| "min": 0.00017486471076749626, | |
| "max": 0.0013649598767580102, | |
| "count": 3 | |
| }, | |
| "Pyramids.Losses.ValueLoss.sum": { | |
| "value": 0.0020983765292099553, | |
| "min": 0.0020983765292099553, | |
| "max": 0.0136495987675801, | |
| "count": 3 | |
| }, | |
| "Pyramids.Policy.LearningRate.mean": { | |
| "value": 5.34945821685e-05, | |
| "min": 5.34945821685e-05, | |
| "max": 0.00013471205509600002, | |
| "count": 3 | |
| }, | |
| "Pyramids.Policy.LearningRate.sum": { | |
| "value": 0.000641934986022, | |
| "min": 0.0005388482203840001, | |
| "max": 0.000982176672608, | |
| "count": 3 | |
| }, | |
| "Pyramids.Policy.Epsilon.mean": { | |
| "value": 0.11783150000000002, | |
| "min": 0.11783150000000002, | |
| "max": 0.144904, | |
| "count": 3 | |
| }, | |
| "Pyramids.Policy.Epsilon.sum": { | |
| "value": 1.4139780000000002, | |
| "min": 0.579616, | |
| "max": 1.4139780000000002, | |
| "count": 3 | |
| }, | |
| "Pyramids.Policy.Beta.mean": { | |
| "value": 0.0017913668500000007, | |
| "min": 0.0017913668500000007, | |
| "max": 0.0044959096, | |
| "count": 3 | |
| }, | |
| "Pyramids.Policy.Beta.sum": { | |
| "value": 0.021496402200000007, | |
| "min": 0.0179836384, | |
| "max": 0.0328064608, | |
| "count": 3 | |
| }, | |
| "Pyramids.Losses.RNDLoss.mean": { | |
| "value": 0.05180659890174866, | |
| "min": 0.05180659890174866, | |
| "max": 0.07853716611862183, | |
| "count": 3 | |
| }, | |
| "Pyramids.Losses.RNDLoss.sum": { | |
| "value": 0.6216791868209839, | |
| "min": 0.3141486644744873, | |
| "max": 0.6216791868209839, | |
| "count": 3 | |
| }, | |
| "Pyramids.Environment.EpisodeLength.mean": { | |
| "value": 999.0, | |
| "min": 998.28125, | |
| "max": 999.0, | |
| "count": 3 | |
| }, | |
| "Pyramids.Environment.EpisodeLength.sum": { | |
| "value": 31968.0, | |
| "min": 15984.0, | |
| "max": 31968.0, | |
| "count": 3 | |
| }, | |
| "Pyramids.Environment.CumulativeReward.mean": { | |
| "value": -0.9999484393385148, | |
| "min": -1.0000000521540642, | |
| "max": -0.8741938006132841, | |
| "count": 3 | |
| }, | |
| "Pyramids.Environment.CumulativeReward.sum": { | |
| "value": -30.99840161949396, | |
| "min": -30.99840161949396, | |
| "max": -16.000000834465027, | |
| "count": 3 | |
| }, | |
| "Pyramids.Policy.ExtrinsicReward.mean": { | |
| "value": -0.9999484393385148, | |
| "min": -1.0000000521540642, | |
| "max": -0.8741938006132841, | |
| "count": 3 | |
| }, | |
| "Pyramids.Policy.ExtrinsicReward.sum": { | |
| "value": -30.99840161949396, | |
| "min": -30.99840161949396, | |
| "max": -16.000000834465027, | |
| "count": 3 | |
| }, | |
| "Pyramids.Policy.RndReward.mean": { | |
| "value": 0.534257584281506, | |
| "min": 0.534257584281506, | |
| "max": 0.8186110365204513, | |
| "count": 3 | |
| }, | |
| "Pyramids.Policy.RndReward.sum": { | |
| "value": 16.56198511272669, | |
| "min": 13.097776584327221, | |
| "max": 19.85759885981679, | |
| "count": 3 | |
| }, | |
| "Pyramids.IsTraining.mean": { | |
| "value": 1.0, | |
| "min": 1.0, | |
| "max": 1.0, | |
| "count": 3 | |
| }, | |
| "Pyramids.IsTraining.sum": { | |
| "value": 1.0, | |
| "min": 1.0, | |
| "max": 1.0, | |
| "count": 3 | |
| } | |
| }, | |
| "metadata": { | |
| "timer_format_version": "0.1.0", | |
| "start_time_seconds": "1782636950", | |
| "python_version": "3.10.12 (main, Jul 5 2023, 18:54:27) [GCC 11.2.0]", | |
| "command_line_arguments": "/usr/local/bin/mlagents-learn ./config/ppo/PyramidsRND.yaml --env=./training-envs-executables/linux/Pyramids/Pyramids --run-id=Pyramids Training --no-graphics --resume", | |
| "mlagents_version": "1.2.0.dev0", | |
| "mlagents_envs_version": "1.2.0.dev0", | |
| "communication_protocol_version": "1.5.0", | |
| "pytorch_version": "2.8.0+cu128", | |
| "numpy_version": "1.23.5", | |
| "end_time_seconds": "1782637165" | |
| }, | |
| "total": 214.87066589799997, | |
| "count": 1, | |
| "self": 0.4798725320001722, | |
| "children": { | |
| "run_training.setup": { | |
| "total": 0.02424729899985323, | |
| "count": 1, | |
| "self": 0.02424729899985323 | |
| }, | |
| "TrainerController.start_learning": { | |
| "total": 214.36654606699994, | |
| "count": 1, | |
| "self": 0.13467098800947497, | |
| "children": { | |
| "TrainerController._reset_env": { | |
| "total": 2.1855458589998307, | |
| "count": 1, | |
| "self": 2.1855458589998307 | |
| }, | |
| "TrainerController.advance": { | |
| "total": 211.81623517799062, | |
| "count": 6266, | |
| "self": 0.1456589550134595, | |
| "children": { | |
| "env_step": { | |
| "total": 150.42087508299892, | |
| "count": 6266, | |
| "self": 134.77207955304357, | |
| "children": { | |
| "SubprocessEnvManager._take_step": { | |
| "total": 15.566905309963659, | |
| "count": 6266, | |
| "self": 0.4703167549682803, | |
| "children": { | |
| "TorchPolicy.evaluate": { | |
| "total": 15.096588554995378, | |
| "count": 6256, | |
| "self": 15.096588554995378 | |
| } | |
| } | |
| }, | |
| "workers": { | |
| "total": 0.08189021999169199, | |
| "count": 6266, | |
| "self": 0.0, | |
| "children": { | |
| "worker_root": { | |
| "total": 213.42091515801076, | |
| "count": 6266, | |
| "is_parallel": true, | |
| "self": 90.6735544820267, | |
| "children": { | |
| "run_training.setup": { | |
| "total": 0.0, | |
| "count": 0, | |
| "is_parallel": true, | |
| "self": 0.0, | |
| "children": { | |
| "steps_from_proto": { | |
| "total": 0.0017912769999384182, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0006375309994837153, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 0.0011537460004547029, | |
| "count": 8, | |
| "is_parallel": true, | |
| "self": 0.0011537460004547029 | |
| } | |
| } | |
| }, | |
| "UnityEnvironment.step": { | |
| "total": 0.05034289000013814, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0005745249995925406, | |
| "children": { | |
| "UnityEnvironment._generate_step_input": { | |
| "total": 0.00046643500036225305, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.00046643500036225305 | |
| }, | |
| "communicator.exchange": { | |
| "total": 0.04747408399998676, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.04747408399998676 | |
| }, | |
| "steps_from_proto": { | |
| "total": 0.0018278460001965868, | |
| "count": 1, | |
| "is_parallel": true, | |
| "self": 0.0003568780002751737, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 0.0014709679999214131, | |
| "count": 8, | |
| "is_parallel": true, | |
| "self": 0.0014709679999214131 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "UnityEnvironment.step": { | |
| "total": 122.74736067598405, | |
| "count": 6265, | |
| "is_parallel": true, | |
| "self": 3.4582741830149644, | |
| "children": { | |
| "UnityEnvironment._generate_step_input": { | |
| "total": 2.3929653799882544, | |
| "count": 6265, | |
| "is_parallel": true, | |
| "self": 2.3929653799882544 | |
| }, | |
| "communicator.exchange": { | |
| "total": 105.69274092498563, | |
| "count": 6265, | |
| "is_parallel": true, | |
| "self": 105.69274092498563 | |
| }, | |
| "steps_from_proto": { | |
| "total": 11.203380187995208, | |
| "count": 6265, | |
| "is_parallel": true, | |
| "self": 2.374361956912253, | |
| "children": { | |
| "_process_rank_one_or_two_observation": { | |
| "total": 8.829018231082955, | |
| "count": 50120, | |
| "is_parallel": true, | |
| "self": 8.829018231082955 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "trainer_advance": { | |
| "total": 61.24970113997824, | |
| "count": 6266, | |
| "self": 0.19105842799581296, | |
| "children": { | |
| "process_trajectory": { | |
| "total": 10.351301033983873, | |
| "count": 6266, | |
| "self": 10.351301033983873 | |
| }, | |
| "_update_policy": { | |
| "total": 50.707341677998556, | |
| "count": 33, | |
| "self": 27.28086230200597, | |
| "children": { | |
| "TorchPPOOptimizer.update": { | |
| "total": 23.426479375992585, | |
| "count": 2283, | |
| "self": 23.426479375992585 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| }, | |
| "trainer_threads": { | |
| "total": 8.679999154992402e-07, | |
| "count": 1, | |
| "self": 8.679999154992402e-07 | |
| }, | |
| "TrainerController._save_models": { | |
| "total": 0.23009317400010332, | |
| "count": 1, | |
| "self": 0.001246918000106234, | |
| "children": { | |
| "RLTrainer._checkpoint": { | |
| "total": 0.22884625599999708, | |
| "count": 1, | |
| "self": 0.22884625599999708 | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } | |
| } |