outsider86 commited on
Commit
7b6ea4b
·
verified ·
1 Parent(s): f0e20bb

Upload folder using huggingface_hub

Browse files
fastumi_pickandplace_qwenPI_329v4/checkpoints/steps_10000_pytorch_model.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9de2530285307659ae80a4a957a109ea970b9f60060b8e41596a45bcef16ad80
3
+ size 12444522384
fastumi_pickandplace_qwenPI_329v4/checkpoints/steps_5000_pytorch_model.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0765328b7819247afcd352b221a31844f9fa33c6ec6ce2c1fdea39f2255f0f3d
3
+ size 12444521025
fastumi_pickandplace_qwenPI_329v4/config.yaml ADDED
@@ -0,0 +1,70 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ datasets:
2
+ vla_data:
3
+ CoT_prompt: Your task is {instruction}. To identify the key objects for your task.
4
+ Locate their bounding boxes in [x1,y1,x2,y2] format.
5
+ data_mix: dynamic-329-v4
6
+ data_root_dir: playground/Datasets/FastUMI
7
+ dataset_py: lerobot_datasets
8
+ per_device_batch_size: 8
9
+ video_backend: torchvision_av
10
+ framework:
11
+ action_model:
12
+ action_dim: 10
13
+ add_pos_embed: true
14
+ diffusion_model_cfg:
15
+ attention_head_dim: 64
16
+ cross_attention_dim: 2048
17
+ dropout: 0.2
18
+ final_dropout: true
19
+ input_embedding_dim: 2048
20
+ interleave_self_attention: true
21
+ norm_type: ada_norm
22
+ num_attention_heads: 32
23
+ num_layers: 36
24
+ output_dim: 1024
25
+ positional_embeddings: null
26
+ future_action_window_size: 15
27
+ max_seq_len: 1024
28
+ noise_beta_alpha: 1.5
29
+ noise_beta_beta: 1.0
30
+ noise_s: 0.999
31
+ num_inference_timesteps: 4
32
+ num_target_vision_tokens: 32
33
+ num_timestep_buckets: 1000
34
+ past_action_window_size: 0
35
+ state_dim: 10
36
+ name: QwenPI
37
+ qwenvl:
38
+ attn_implementation: flash_attention_2
39
+ base_vlm: playground/Pretrained_models/Qwen2.5-VL-3B-Instruct-Action
40
+ num_vl_layers: 36
41
+ vl_hidden_dim: 2048
42
+ output_dir: ./results/Checkpoints/fastumi_pickandplace_qwenPI_329v4
43
+ run_id: fastumi_pickandplace_qwenPI_329v4
44
+ run_root_dir: ./results/Checkpoints
45
+ seed: 42
46
+ trainer:
47
+ eval_interval: 100
48
+ freeze_modules: null
49
+ gradient_accumulation_steps: 1
50
+ gradient_clipping: 1.0
51
+ is_resume: true
52
+ learning_rate:
53
+ action_model: 0.0001
54
+ base: 2.5e-05
55
+ qwen_vl_interface: 1.0e-05
56
+ logging_frequency: 50
57
+ lr_scheduler_type: cosine_with_min_lr
58
+ max_train_steps: 10000
59
+ num_warmup_steps: 5000
60
+ optimizer:
61
+ betas:
62
+ - 0.9
63
+ - 0.95
64
+ eps: 1.0e-08
65
+ weight_decay: 1.0e-08
66
+ save_interval: 5000
67
+ scheduler_specific_kwargs:
68
+ min_lr: 1.0e-06
69
+ wandb_entity: 2200011093-peking-university
70
+ wandb_project: starVLA_FastUMI_dynamic_1
fastumi_pickandplace_qwenPI_329v4/dataset_statistics.json ADDED
@@ -0,0 +1,178 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "new_embodiment": {
3
+ "action": {
4
+ "mean": [
5
+ 8.114812226267532e-05,
6
+ 0.0011773157166317105,
7
+ -0.0008882488473318517,
8
+ 0.9999980926513672,
9
+ 0.0004620188265107572,
10
+ -5.512892585102236e-06,
11
+ -0.00045027409214526415,
12
+ 0.9999972581863403,
13
+ 0.0008542306604795158,
14
+ 0.5120490193367004
15
+ ],
16
+ "std": [
17
+ 0.002003167988732457,
18
+ 0.0030834961216896772,
19
+ 0.00332102132961154,
20
+ 5.601261758867448e-05,
21
+ 0.00517509737983346,
22
+ 0.006088857538998127,
23
+ 0.00517628388479352,
24
+ 7.67255479378806e-05,
25
+ 0.007195365149527788,
26
+ 0.4998091757297516
27
+ ],
28
+ "max": [
29
+ 0.008002996444702148,
30
+ 0.01195499300956726,
31
+ 0.009996004402637482,
32
+ 1.0000001192092896,
33
+ 0.03673427179455757,
34
+ 0.05374615639448166,
35
+ 0.02689175307750702,
36
+ 1.0000001192092896,
37
+ 0.06838423758745193,
38
+ 1.0
39
+ ],
40
+ "min": [
41
+ -0.00818699598312378,
42
+ -0.008041977882385254,
43
+ -0.010141998529434204,
44
+ 0.9985396265983582,
45
+ -0.026854006573557854,
46
+ -0.04442450404167175,
47
+ -0.03635682165622711,
48
+ 0.997472882270813,
49
+ -0.06104228273034096,
50
+ 0.0
51
+ ],
52
+ "q01": [
53
+ -0.00486600399017334,
54
+ -0.00501847505569458,
55
+ -0.007628242671489716,
56
+ 0.9997773110866547,
57
+ -0.012261943705379964,
58
+ -0.015310721378773451,
59
+ -0.013358078151941299,
60
+ 0.9997043466567993,
61
+ -0.01697732284665108,
62
+ 0.0
63
+ ],
64
+ "q99": [
65
+ 0.005302621722221377,
66
+ 0.009075625538825991,
67
+ 0.007293619811534884,
68
+ 0.9999998211860657,
69
+ 0.013378759231418375,
70
+ 0.016290122829377692,
71
+ 0.01225539410486818,
72
+ 0.9999997615814209,
73
+ 0.020962502509355545,
74
+ 1.0
75
+ ],
76
+ "mask": [
77
+ true,
78
+ true,
79
+ true,
80
+ true,
81
+ true,
82
+ true,
83
+ true,
84
+ true,
85
+ true,
86
+ false
87
+ ],
88
+ "norm_modes": [
89
+ "min_max",
90
+ "min_max",
91
+ "min_max",
92
+ "none",
93
+ "none",
94
+ "none",
95
+ "none",
96
+ "none",
97
+ "none",
98
+ "binary"
99
+ ]
100
+ },
101
+ "state": {
102
+ "mean": [
103
+ 0.27271032333374023,
104
+ -0.446323037147522,
105
+ 0.1041000559926033,
106
+ -0.59544438123703,
107
+ 0.37788620591163635,
108
+ -0.6611562967300415,
109
+ 0.667895495891571,
110
+ 0.31084170937538147,
111
+ -0.42228883504867554,
112
+ 0.5119883418083191
113
+ ],
114
+ "std": [
115
+ 0.043663933873176575,
116
+ 0.0719655305147171,
117
+ 0.06168781593441963,
118
+ 0.05646670609712601,
119
+ 0.24672336876392365,
120
+ 0.0382271967828272,
121
+ 0.4340234696865082,
122
+ 0.08751874417066574,
123
+ 0.28798049688339233,
124
+ 0.49981093406677246
125
+ ],
126
+ "max": [
127
+ 0.4219360053539276,
128
+ -0.17201299965381622,
129
+ 0.33353498578071594,
130
+ -0.34213387966156006,
131
+ 0.6040946245193481,
132
+ -0.5470927953720093,
133
+ 0.9302739500999451,
134
+ 0.70821213722229,
135
+ 0.611838161945343,
136
+ 1.0
137
+ ],
138
+ "min": [
139
+ 0.12113799899816513,
140
+ -0.5658149719238281,
141
+ 0.011877999641001225,
142
+ -0.745457112789154,
143
+ -0.5488103032112122,
144
+ -0.7786615490913391,
145
+ -0.8129586577415466,
146
+ -0.03098832629621029,
147
+ -0.6885839700698853,
148
+ 0.0
149
+ ],
150
+ "q01": [
151
+ 0.15309083521366118,
152
+ -0.5302998530864715,
153
+ 0.01954999938607216,
154
+ -0.7032991063594818,
155
+ -0.47749382197856904,
156
+ -0.7447073876857757,
157
+ -0.7802488780021668,
158
+ 0.07434731468558312,
159
+ -0.6129826271533966,
160
+ 0.0
161
+ ],
162
+ "q99": [
163
+ 0.38105886042118076,
164
+ -0.27944067239761344,
165
+ 0.27689720153808595,
166
+ -0.4402932196855544,
167
+ 0.5639739322662355,
168
+ -0.5716438913345336,
169
+ 0.8863004243373871,
170
+ 0.5411446142196659,
171
+ 0.5678422379493714,
172
+ 1.0
173
+ ]
174
+ },
175
+ "num_transitions": 65939,
176
+ "num_trajectories": 525
177
+ }
178
+ }
fastumi_pickandplace_qwenPI_329v4/final_model/pytorch_model.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9100766b6882b860c456af520cf4b9ec2281826a24b29db27f01b0e90275bb38
3
+ size 12444501852
fastumi_pickandplace_qwenPI_329v4/summary.jsonl ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ {"steps": 5000}
2
+ {"steps": 10000}